@inproceedings{bb245800,
AUTHOR = "Yang, K.C. and Deng, J.K. and An, X. and Li, J.W. and Feng, Z. and Guo, J. and Yang, J. and Liu, T.L.",
TITLE = "ALIP: Adaptive Language-Image Pre-training with Synthetic Caption",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "2910-2919",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240709"}
@inproceedings{bb245801,
AUTHOR = "Yang, Y.F. and Huang, W.Q. and Wei, Y.X. and Peng, H. and Jiang, X.Y. and Jiang, H.Q. and Wei, F.Y. and Wang, Y. and Hu, H. and Qiu, L. and Yang, Y.Q.",
TITLE = "Attentive Mask CLIP",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "2759-2769",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240710"}
@inproceedings{bb245802,
AUTHOR = "Vinker, Y. and Alaluf, Y. and Cohen Or, D. and Shamir, A.",
TITLE = "CLIPascene: Scene Sketching with Different Types and Levels of
Abstraction",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "4123-4133",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240711"}
@inproceedings{bb245803,
AUTHOR = "Wei, Y.X. and Hu, H. and Xie, Z. and Liu, Z. and Zhang, Z. and Cao, Y. and Bao, J.M. and Chen, D. and Guo, B.",
TITLE = "Improving CLIP Fine-tuning Performance",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "5416-5426",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240712"}
@inproceedings{bb245804,
AUTHOR = "Maniparambil, M. and Vorster, C. and Molloy, D. and Murphy, N. and McGuinness, K. and O'Connor, N.E.",
TITLE = "Enhancing CLIP with GPT-4: Harnessing Visual Descriptions as Prompts",
BOOKTITLE = MMFM23,
YEAR = "2023",
PAGES = "262-271",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240713"}
@inproceedings{bb245805,
AUTHOR = "Zheng, X. and Huang, X.S. and Mei, G.F. and Hou, Y.N. and Lyu, Z.Y. and Dai, B. and Ouyang, W.L. and Gong, Y.S.",
TITLE = "Point Cloud Pre-Training with Diffusion Models",
BOOKTITLE = CVPR24,
YEAR = "2024",
PAGES = "22935-22945",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240714"}
@inproceedings{bb245806,
AUTHOR = "Huang, T.Y. and Dong, B. and Yang, Y.H. and Huang, X.S. and Lau, R.W.H. and Ouyang, W.L. and Zuo, W.M.",
TITLE = "CLIP2Point: Transfer CLIP to Point Cloud Classification with
Image-Depth Pre-Training",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "22100-22110",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240715"}
@inproceedings{bb245807,
AUTHOR = "Wu, K. and Peng, H.W. and Zhou, Z.H. and Xiao, B. and Liu, M.C. and Yuan, L. and Xuan, H. and Valenzuela, M. and Chen, X.S. and Wang, X.G. and Chao, H.Y. and Hu, H.",
TITLE = "TinyCLIP: CLIP Distillation via Affinity Mimicking and Weight
Inheritance",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "21913-21923",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240716"}
@inproceedings{bb245808,
AUTHOR = "Deng, X.C. and Shi, H. and Huang, R.H. and Li, C.L. and Xu, H. and Han, J.H. and Kwok, J. and Zhao, S. and Zhang, W. and Liang, X.D.",
TITLE = "GrowCLIP: Data-aware Automatic Model Growing for Large-scale
Contrastive Language-Image Pre-training",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "22121-22132",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240717"}
@inproceedings{bb245809,
AUTHOR = "Ranasinghe, K. and McKinzie, B. and Ravi, S. and Yang, Y.F. and Toshev, A. and Shlens, J.",
TITLE = "Perceptual Grouping in Contrastive Vision-Language Models",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "5548-5561",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240718"}
@inproceedings{bb245810,
AUTHOR = "Shao, B. and Liu, J.Z. and Pei, R. and Xu, S. and Dai, P. and Lu, J.W. and Li, W.M. and Yan, Y.L.",
TITLE = "HiVLP: Hierarchical Interactive Video-Language Pre-Training",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "13710-13720",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240719"}
@inproceedings{bb245811,
AUTHOR = "Ali, M. and Khan, S.",
TITLE = "CLIP-Decoder: ZeroShot Multilabel Classification using Multimodal
CLIP Aligned Representations",
BOOKTITLE = VLAR23,
YEAR = "2023",
PAGES = "4677-4681",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240720"}
@inproceedings{bb245812,
AUTHOR = "Singha, M. and Pal, H. and Jha, A. and Banerjee, B.",
TITLE = "AD-CLIP: Adapting Domains in Prompt Space Using CLIP",
BOOKTITLE = OutDistri23,
YEAR = "2023",
PAGES = "4357-4366",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240721"}
@inproceedings{bb245813,
AUTHOR = "Zhang, J. and Dong, R. and Ma, K.",
TITLE = "CLIP-FO3D:
Learning Free Open-world 3D Scene Representations from 2D Dense CLIP",
BOOKTITLE = OpenSUN3D,
PAGES = "2040-2051",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240722"}
@inproceedings{bb245814,
AUTHOR = "Auty, D. and Mikolajczyk, K.",
TITLE = "Learning to Prompt CLIP for Monocular Depth Estimation:
Exploring the Limits of Human Language",
BOOKTITLE = OpenSUN3D,
PAGES = "2031-2049",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240723"}
@inproceedings{bb245815,
AUTHOR = "Xu, X. and Xiong, T.Y. and Ding, Z. and Tu, Z.W.",
TITLE = "MasQCLIP for Open-Vocabulary Universal Image Segmentation",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "887-898",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240724"}
@inproceedings{bb245816,
AUTHOR = "Wang, H.L. and Li, Y. and Yao, H.F. and Li, X.M.",
TITLE = "CLIPN for Zero-Shot OOD Detection: Teaching CLIP to Say No",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "1802-1812",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240725"}
@inproceedings{bb245817,
AUTHOR = "Zhu, X.Y. and Zhang, R.R. and He, B. and Zhou, A. and Wang, D. and Zhao, B. and Gao, P.",
TITLE = "Not All Features Matter:
Enhancing Few-shot CLIP with Adaptive Prior Refinement",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "2605-2615",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240726"}
@inproceedings{bb245818,
AUTHOR = "Paiss, R. and Ephrat, A. and Tov, O. and Zada, S. and Mosseri, I. and Irani, M. and Dekel, T.",
TITLE = "Teaching CLIP to Count to Ten",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "3147-3157",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240727"}
@inproceedings{bb245819,
AUTHOR = "Zhu, X.Y. and Zhang, R.R. and He, B. and Guo, Z.Y. and Zeng, Z. and Qin, Z. and Zhang, S.H. and Gao, P.",
TITLE = "PointCLIP V2: Prompting CLIP and GPT for Powerful 3D Open-world
Learning",
BOOKTITLE = ICCV23,
YEAR = "2023",
PAGES = "2639-2650",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240728"}
@inproceedings{bb245820,
AUTHOR = "Yuan, M. and Lv, N.N. and Xie, Y.F. and Lu, F.X. and Zhan, K.",
TITLE = "CLIP-FG: Selecting Discriminative Image Patches by Contrastive
Language-Image Pre-Training for Fine-Grained Image Classification",
BOOKTITLE = ICIP23,
YEAR = "2023",
PAGES = "560-564",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240729"}
@inproceedings{bb245821,
AUTHOR = "Zeng, Z.Y. and Ge, Y.Y. and Liu, X.H. and Chen, B. and Luo, P. and Xia, S.T. and Ge, Y.X.",
TITLE = "Learning Transferable Spatiotemporal Representations from Natural
Script Knowledge",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "23079-23089",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240730"}
@inproceedings{bb245822,
AUTHOR = "Wang, J.P. and Ge, Y.X. and Yan, R. and Ge, Y.Y. and Lin, K.Q.H. and Tsutsui, S. and Lin, X.D. and Cai, G. and Wu, J.P. and Shan, Y. and Qie, X. and Shou, M.Z.",
TITLE = "All in One: Exploring Unified Video-Language Pre-Training",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "6598-6608",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240731"}
@inproceedings{bb245823,
AUTHOR = "Ramrakhya, R. and Batra, D. and Wijmans, E. and Das, A.",
TITLE = "PIRLNav: Pretraining with Imitation and RL Finetuning for OBJECTNAV",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "17896-17906",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240732"}
@inproceedings{bb245824,
AUTHOR = "Lin, X.D. and Tiwari, S. and Huang, S.Y. and Li, M. and Shou, M.Z. and Ji, H. and Chang, S.F.",
TITLE = "Towards Fast Adaptation of Pretrained Contrastive Models for
Multi-channel Video-Language Retrieval",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "14846-14855",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240733"}
@inproceedings{bb245825,
AUTHOR = "Wang, H.C. and Du, X.D. and Li, J.H. and Yeh, R.A. and Shakhnarovich, G.",
TITLE = "Score Jacobian Chaining: Lifting Pretrained 2D Diffusion Models for
3D Generation",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "12619-12629",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240734"}
@inproceedings{bb245826,
AUTHOR = "Luo, Y.X. and Ji, J.Y. and Chen, X.F. and Zhang, Y.X. and Ren, T. and Luo, G.",
TITLE = "APL: Anchor-based Prompt Learning for One-stage Weakly Supervised
Referring Expression Comprehension",
BOOKTITLE = ECCV24,
YEAR = "2024",
PAGES = "XIII: 198-215",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240735"}
@inproceedings{bb245827,
AUTHOR = "Jin, L. and Luo, G. and Zhou, Y.Y. and Sun, X.S. and Jiang, G.N. and Shu, A. and Ji, R.R.",
TITLE = "RefCLIP: A Universal Teacher for Weakly Supervised Referring
Expression Comprehension",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "01-10",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240736"}
@inproceedings{bb245828,
AUTHOR = "Saito, K. and Sohn, K. and Zhang, X. and Li, C.L. and Lee, C.Y. and Saenko, K. and Pfister, T.",
TITLE = "Prefix Conditioning Unifies Language and Label Supervision",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "2861-2870",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240737"}
@inproceedings{bb245829,
AUTHOR = "Park, J. and Han, B.H.",
TITLE = "Multi-Modal Representation Learning with Text-Driven Soft Masks",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "2798-2807",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240738"}
@inproceedings{bb245830,
AUTHOR = "Jin, Z. and Hayat, M. and Yang, Y.W. and Guo, Y.L. and Lei, Y.J.",
TITLE = "Context-aware Alignment and Mutual Masking for 3D-Language
Pre-training",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "10984-10994",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240739"}
@inproceedings{bb245831,
AUTHOR = "Guo, Z.X. and Dong, B. and Ji, Z.L. and Bai, J.F. and Guo, Y.W. and Zuo, W.M.",
TITLE = "Texts as Images in Prompt Tuning for Multi-Label Image Recognition",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "2808-2817",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240740"}
@inproceedings{bb245832,
AUTHOR = "Cherti, M. and Beaumont, R. and Wightman, R. and Wortsman, M. and Ilharco, G. and Gordon, C. and Schuhmann, C. and Schmidt, L. and Jitsev, J.",
TITLE = "Reproducible Scaling Laws for Contrastive Language-Image Learning",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "2818-2829",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240741"}
@inproceedings{bb245833,
AUTHOR = "Lei, J. and Li, L.J. and Zhou, L. and Gan, Z. and Berg, T.L. and Bansal, M. and Liu, J.J.",
TITLE = "Less is More:
CLIPBERT for Video-and-Language Learning via Sparse Sampling",
BOOKTITLE = CVPR21,
YEAR = "2021",
PAGES = "7327-7337",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240742"}
@inproceedings{bb245834,
AUTHOR = "Zhou, J.H. and Dong, L. and Gan, Z. and Wang, L.J. and Wei, F.",
TITLE = "Non-Contrastive Learning Meets Language-Image Pre-Training",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "11028-11038",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240743"}
@inproceedings{bb245835,
AUTHOR = "Hu, Z. and Iscen, A. and Sun, C. and Wang, Z.R. and Chang, K.W. and Sun, Y.Z. and Schmid, C. and Ross, D.A. and Fathi, A.",
TITLE = "Reveal: Retrieval-Augmented Visual-Language Pre-Training with
Multi-Source Multimodal Knowledge Memory",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "23369-23379",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240744"}
@inproceedings{bb245836,
AUTHOR = "Li, Y.H. and Fan, H.Q. and Hu, R.H. and Feichtenhofer, C. and He, K.M.",
TITLE = "Scaling Language-Image Pre-Training via Masking",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "23390-23400",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240745"}
@inproceedings{bb245837,
AUTHOR = "Jin, P. and Huang, J. and Xiong, P.F. and Tian, S.X. and Liu, C. and Ji, X.Y. and Yuan, L. and Chen, J.",
TITLE = "Video-Text as Game Players: Hierarchical Banzhaf Interaction for
Cross-Modal Representation Learning",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "2472-2482",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240746"}
@inproceedings{bb245838,
AUTHOR = "Ye, S.Q. and Xie, Y.J. and Chen, D.D. and Xu, Y. and Yuan, L. and Zhu, C.G. and Liao, J.",
TITLE = "Improving Commonsense in Vision-Language Models via Knowledge Graph
Riddles",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "2634-2645",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240747"}
@inproceedings{bb245839,
AUTHOR = "Li, H. and Zhu, J.G. and Jiang, X.H. and Zhu, X.Z. and Li, H.S. and Yuan, C. and Wang, X.H. and Qiao, Y. and Wang, X.G. and Wang, W.H. and Dai, J.F.",
TITLE = "Uni-Perceiver v2: A Generalist Model for Large-Scale Vision and
Vision-Language Tasks",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "2691-2700",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240748"}
@inproceedings{bb245840,
AUTHOR = "Wu, W.H. and Wang, X.H. and Luo, H.P. and Wang, J.D. and Yang, Y. and Ouyang, W.L.",
TITLE = "Bidirectional Cross-Modal Knowledge Exploration for Video Recognition
with Pre-trained Vision-Language Models",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "6620-6630",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240749"}
@inproceedings{bb245841,
AUTHOR = "Seth, A. and Hemani, M. and Agarwal, C.",
TITLE = "DeAR: Debiasing Vision-Language Models with Additive Residuals",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "6820-6829",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240750"}
@inproceedings{bb245842,
AUTHOR = "Radenovic, F. and Dubey, A. and Kadian, A. and Mihaylov, T. and Vandenhende, S. and Patel, Y. and Wen, Y. and Ramanathan, V. and Mahajan, D.",
TITLE = "Filtering, Distillation, and Hard Negatives for Vision-Language
Pre-Training",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "6967-6977",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240751"}
@inproceedings{bb245843,
AUTHOR = "Yu, T. and Lu, Z. and Jin, X. and Chen, Z.B. and Wang, X.C.",
TITLE = "Task Residual for Tuning Vision-Language Models",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "10899-10909",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240752"}
@inproceedings{bb245844,
AUTHOR = "Yin, D. and Gao, F. and Thattai, G. and Johnston, M. and Chang, K.W.",
TITLE = "GIVL: Improving Geographical Inclusivity of Vision-Language Models
with Pre-Training Methods",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "10951-10961",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240753"}
@inproceedings{bb245845,
AUTHOR = "Gao, C. and Peng, X.Y. and Yan, M. and Wang, H. and Yang, L.R. and Ren, H.B. and Li, H.S. and Liu, S.",
TITLE = "Adaptive Zone-aware Hierarchical Planner for Vision-Language
Navigation",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "14911-14920",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240754"}
@inproceedings{bb245846,
AUTHOR = "Yeh, C.H. and Russell, B. and Sivic, J. and Heilbron, F.C. and Jenni, S.",
TITLE = "Meta-Personalizing Vision-Language Models to Find Named Instances in
Video",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "19123-19132",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240755"}
@inproceedings{bb245847,
AUTHOR = "Gou, Y.H. and Ko, T. and Yang, H. and Kwok, J. and Zhang, Y. and Wang, M.X.",
TITLE = "Leveraging per Image-Token Consistency for Vision-Language
Pre-Training",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "19155-19164",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240756"}
@inproceedings{bb245848,
AUTHOR = "Wang, S.J. and Chang, J.L. and Li, H.J. and Wang, Z.H. and Ouyang, W.L. and Tian, Q.",
TITLE = "Open-Set Fine-Grained Retrieval via Prompting Vision-Language
Evaluator",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "19381-19391",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240757"}
@inproceedings{bb245849,
AUTHOR = "Cheng, F. and Wang, X.Z. and Lei, J. and Crandall, D. and Bansal, M. and Bertasius, G.",
TITLE = "VindLU: A Recipe for Effective Video-and-Language Pretraining",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "10739-10750",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240758"}
@inproceedings{bb245850,
AUTHOR = "Zhou, H.L. and Martin Martin, R. and Kapadia, M. and Savarese, S. and Niebles, J.C.",
TITLE = "Procedure-Aware Pretraining for Instructional Video Understanding",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "10727-10738",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240759"}
@inproceedings{bb245851,
AUTHOR = "Yang, A. and Nagrani, A. and Seo, P.H. and Miech, A. and Pont Tuset, J. and Laptev, I. and Sivic, J. and Schmid, C.",
TITLE = "Vid2Seq: Large-Scale Pretraining of a Visual Language Model for Dense
Video Captioning",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "10714-10726",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240760"}
@inproceedings{bb245852,
AUTHOR = "Alper, M. and Fiman, M. and Averbuch Elor, H.",
TITLE = "Is BERT Blind? Exploring the Effect of Vision-and-Language
Pretraining on Visual Language Understanding",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "6778-6788",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240761"}
@inproceedings{bb245853,
AUTHOR = "Liu, M.Y. and Jiang, J. and Zhu, C. and Yin, X.C.",
TITLE = "VLPD: Context-Aware Pedestrian Detection via Vision-Language Semantic
Self-Supervision",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "6662-6671",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240762"}
@inproceedings{bb245854,
AUTHOR = "Wei, Y.X. and Cao, Y. and Zhang, Z. and Peng, H. and Yao, Z.L. and Xie, Z. and Hu, H. and Guo, B.",
TITLE = "iCLIP: Bridging Image Classification and Contrastive Language-Image
Pre-training for Visual Recognition",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "2776-2786",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240763"}
@inproceedings{bb245855,
AUTHOR = "Hyung, J. and Hwang, S. and Kim, D. and Lee, H. and Choo, J.",
TITLE = "Local 3D Editing via 3D Distillation of CLIP Knowledge",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "12674-12684",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240764"}
@inproceedings{bb245856,
AUTHOR = "Mu, N. and Kirillov, A. and Wagner, D. and Xie, S.",
TITLE = "SLIP: Self-supervision Meets Language-Image Pre-training",
BOOKTITLE = ECCV22,
YEAR = "2022",
PAGES = "XXVI:529-544",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240765"}
@inproceedings{bb245857,
AUTHOR = "Crowson, K. and Biderman, S. and Kornis, D. and Stander, D. and Hallahan, E. and Castricato, L. and Raff, E.",
TITLE = "VQGAN-CLIP: Open Domain Image Generation and Editing with Natural
Language Guidance",
BOOKTITLE = ECCV22,
YEAR = "2022",
PAGES = "XXXVII:88-105",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240766"}
@inproceedings{bb245858,
AUTHOR = "Wu, X.S. and Zhu, F. and Zhao, R. and Li, H.S.",
TITLE = "CORA: Adapting CLIP for Open-Vocabulary Detection with Region
Prompting and Anchor Pre-Matching",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "7031-7040",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240767"}
@inproceedings{bb245859,
AUTHOR = "Dong, X.Y. and Bao, J.M. and Zheng, Y.L. and Zhang, T. and Chen, D.D. and Yang, H. and Zeng, M. and Zhang, W.M. and Yuan, L. and Chen, D. and Wen, F. and Yu, N.H.",
TITLE = "MaskCLIP: Masked Self-Distillation Advances Contrastive
Language-Image Pretraining",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "10995-11005",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240768"}
@inproceedings{bb245860,
AUTHOR = "Xie, C.W. and Sun, S.Y. and Xiong, X. and Zheng, Y. and Zhao, D.L. and Zhou, J.R.",
TITLE = "RA-CLIP: Retrieval Augmented Contrastive Language-Image Pre-Training",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "19265-19274",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240769"}
@inproceedings{bb245861,
AUTHOR = "Chen, P.J. and Li, Q. and Biaz, S. and Bui, T. and Nguyen, A.",
TITLE = "gScoreCAM: What Objects Is CLIP Looking At?",
BOOKTITLE = ACCV22,
YEAR = "2022",
PAGES = "IV:588-604",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240770"}
@inproceedings{bb245862,
AUTHOR = "Wang, R. and Duan, X.Y. and Kang, G.L. and Liu, J.Z. and Lin, S.H. and Xu, S. and Lv, J. and Zhang, B.C.",
TITLE = "AttriCLIP: A Non-Incremental Learner for Incremental Knowledge Learning",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "3654-3663",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240771"}
@inproceedings{bb245863,
AUTHOR = "Rasheed, H. and Khattak, M.U. and Maaz, M. and Khan, S. and Khan, F.S.",
TITLE = "Fine-tuned CLIP Models are Efficient Video Learners",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "6545-6554",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240772"}
@inproceedings{bb245864,
AUTHOR = "Liu, R. and Huang, J.J. and Li, G. and Feng, J.S. and Wu, X.L. and Li, T.H.",
TITLE = "Revisiting Temporal Modeling for CLIP-Based Image-to-Video Knowledge
Transferring",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "6555-6564",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240773"}
@inproceedings{bb245865,
AUTHOR = "Tschannen, M. and Mustafa, B. and Houlsby, N.",
TITLE = "CLIPPO: Image-and-Language Understanding from Pixels Only",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "11006-11017",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240774"}
@inproceedings{bb245866,
AUTHOR = "Zhou, Z.Q. and Lei, Y.J. and Zhang, B. and Liu, L.Q. and Liu, Y.F.",
TITLE = "ZegCLIP: Towards Adapting CLIP for Zero-shot Semantic Segmentation",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "11175-11185",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240775"}
@inproceedings{bb245867,
AUTHOR = "He, W.B. and Jamonnak, S. and Gou, L. and Ren, L.",
TITLE = "CLIP-S4: Language-Guided Self-Supervised Semantic Segmentation",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "11207-11216",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240776"}
@inproceedings{bb245868,
AUTHOR = "Huang, Z.X. and Jampani, V. and Thai, A. and Li, Y.Z. and Stojanov, S. and Rehg, J.M.",
TITLE = "ShapeClipper: Scalable 3D Shape Learning from Single-View Images via
Geometric and CLIP-Based Consistency",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "12912-12922",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240777"}
@inproceedings{bb245869,
AUTHOR = "Zeng, Y.H. and Jiang, C.H. and Mao, J.G. and Han, J.H. and Ye, C.Q. and Huang, Q.Q. and Yeung, D.Y. and Yang, Z. and Liang, X.D. and Xu, H.",
TITLE = "CLIP2: Contrastive Language-Image-Point Pretraining from Real-World
Point Cloud Data",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "15244-15253",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240778"}
@inproceedings{bb245870,
AUTHOR = "Lin, Y.Q. and Chen, M.H. and Wang, W.X. and Wu, B. and Li, K. and Lin, B.B. and Liu, H.F. and He, X.F.",
TITLE = "CLIP is Also an Efficient Segmenter: A Text-Driven Approach for
Weakly Supervised Semantic Segmentation",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "15305-15314",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240779"}
@inproceedings{bb245871,
AUTHOR = "Sanghi, A. and Fu, R. and Liu, V. and Willis, K.D.D. and Shayani, H. and Khasahmadi, A.H. and Sridhar, S. and Ritchie, D.",
TITLE = "CLIP-Sculptor: Zero-Shot Generation of High-Fidelity and Diverse
Shapes from Natural Language",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "18339-18348",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240780"}
@inproceedings{bb245872,
AUTHOR = "Pei, R.J. and Liu, J.Z. and Li, W.M. and Shao, B. and Xu, S. and Dai, P. and Lu, J.W. and Yan, Y.",
TITLE = "CLIPPING: Distilling CLIP-Based Models with a Student Base for
Video-Language Retrieval",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "18983-18992",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240781"}
@inproceedings{bb245873,
AUTHOR = "Jeong, J. and Zou, Y. and Kim, T. and Zhang, D.Q. and Ravichandran, A. and Dabeer, O.",
TITLE = "WinCLIP: Zero-/Few-Shot Anomaly Classification and Segmentation",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "19606-19616",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240782"}
@inproceedings{bb245874,
AUTHOR = "Shamshad, F. and Naseer, M. and Nandakumar, K.",
TITLE = "CLIP2Protect: Protecting Facial Privacy Using Text-Guided Makeup via
Adversarial Latent Search",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "20595-20605",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240783"}
@inproceedings{bb245875,
AUTHOR = "Chen, Y.H. and Qi, X. and Wang, J.A. and Zhang, L.",
TITLE = "DisCo-CLIP: A Distributed Contrastive Loss for Memory Efficient CLIP
Training",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "22648-22657",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240784"}
@inproceedings{bb245876,
AUTHOR = "Wasim, S.T. and Naseer, M. and Khan, S. and Khan, F.S. and Shah, M.",
TITLE = "Vita-CLIP: Video and text adaptive CLIP via Multimodal Prompting",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "23034-23044",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240785"}
@inproceedings{bb245877,
AUTHOR = "Parelli, M. and Delitzas, A. and Hars, N. and Vlassis, G. and Anagnostidis, S. and Bachmann, G. and Hofmann, T.",
TITLE = "CLIP-Guided Vision-Language Pre-training for Question Answering in 3D
Scenes",
BOOKTITLE = ODRUM23,
YEAR = "2023",
PAGES = "5607-5612",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240786"}
@inproceedings{bb245878,
AUTHOR = "Ning, S. and Qiu, L.T. and Liu, Y.F. and He, X.M.",
TITLE = "HOICLIP: Efficient Knowledge Transfer for HOI Detection with
Vision-Language Models",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "23507-23517",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240787"}
@inproceedings{bb245879,
AUTHOR = "Yao, L.W. and Han, J.H. and Liang, X.D. and Xu, D. and Zhang, W. and Li, Z.G. and Xu, H.",
TITLE = "DetCLIPv2: Scalable Open-Vocabulary Object Detection Pre-training via
Word-Region Alignment",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "23497-23506",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240788"}
@inproceedings{bb245880,
AUTHOR = "Singha, M. and Jha, A. and Solanki, B. and Bose, S. and Banerjee, B.",
TITLE = "APPLeNet: Visual Attention Parameterized Prompt Learning for Few-Shot
Remote Sensing Image Generalization using CLIP",
BOOKTITLE = EarthVision23,
YEAR = "2023",
PAGES = "2024-2034",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240789"}
@inproceedings{bb245881,
AUTHOR = "Gannamaneni, S.S. and Sadaghiani, A. and Rao, R.P. and Mock, M. and Akila, M.",
TITLE = "Investigating CLIP Performance for Meta-data Generation in AD
Datasets",
BOOKTITLE = SAIAD23,
YEAR = "2023",
PAGES = "3840-3850",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240790"}
@inproceedings{bb245882,
AUTHOR = "Chen, R.N. and Liu, Y.Q. and Kong, L.D. and Zhu, X.G. and Ma, Y.X. and Li, Y.K. and Hou, Y.N. and Qiao, Y. and Wang, W.P.",
TITLE = "CLIP2Scene: Towards Label-efficient 3D Scene Understanding by CLIP",
BOOKTITLE = CVPR23,
YEAR = "2023",
PAGES = "7020-7030",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240791"}
@inproceedings{bb245883,
AUTHOR = "Ni, B.L. and Peng, H.W. and Chen, M.H. and Zhang, S.Y. and Meng, G.F. and Fu, J.L. and Xiang, S.M. and Ling, H.B.",
TITLE = "Expanding Language-Image Pretrained Models for General Video
Recognition",
BOOKTITLE = ECCV22,
YEAR = "2022",
PAGES = "IV:1-18",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240792"}
@inproceedings{bb245884,
AUTHOR = "Zhang, R.R. and Zhang, W. and Fang, R.Y. and Gao, P. and Li, K.C. and Dai, J.F. and Qiao, Y. and Li, H.S.",
TITLE = "Tip-Adapter: Training-Free Adaption of CLIP for Few-Shot Classification",
BOOKTITLE = ECCV22,
YEAR = "2022",
PAGES = "XXXV:493-510",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240793"}
@inproceedings{bb245885,
AUTHOR = "Yang, J. and Duan, J.L. and Tran, S. and Xu, Y. and Chanda, S. and Chen, L.Q. and Zeng, B. and Chilimbi, T. and Huang, J.Z.",
TITLE = "Vision-Language Pre-Training with Triple Contrastive Learning",
BOOKTITLE = CVPR22,
YEAR = "2022",
PAGES = "15650-15659",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240794"}
@inproceedings{bb245886,
AUTHOR = "Guo, X.Y. and Duan, J.L. and Kuo, C.C.J. and Gichoya, J.W. and Banerjee, I.",
TITLE = "Augmenting Vision Language Pretraining by Learning Codebook with
Visual Semantics",
BOOKTITLE = "ICPR22",
YEAR = "2022",
PAGES = "4779-4785",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240795"}
@inproceedings{bb245887,
AUTHOR = "Zhou, C. and Loy, C.C. and Dai, B.",
TITLE = "Extract Free Dense Labels from CLIP",
BOOKTITLE = ECCV22,
YEAR = "2022",
PAGES = "XXVIII:696-712",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240796"}
@inproceedings{bb245888,
AUTHOR = "Lin, Z. and Geng, S.J. and Zhang, R.R. and Gao, P. and de Melo, G. and Wang, X.G. and Dai, J.F. and Qiao, Y. and Li, H.S.",
TITLE = "Frozen CLIP Models are Efficient Video Learners",
BOOKTITLE = ECCV22,
YEAR = "2022",
PAGES = "XXXV:388-404",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240797"}
@inproceedings{bb245889,
AUTHOR = "Rao, Y.M. and Zhao, W.L. and Chen, G.Y. and Tang, Y.S. and Zhu, Z. and Huang, G. and Zhou, J. and Lu, J.W.",
TITLE = "DenseCLIP: Language-Guided Dense Prediction with Context-Aware
Prompting",
BOOKTITLE = CVPR22,
YEAR = "2022",
PAGES = "18061-18070",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240798"}
@inproceedings{bb245890,
AUTHOR = "Kwon, G. and Ye, J.C.",
TITLE = "CLIPstyler: Image Style Transfer with a Single Text Condition",
BOOKTITLE = CVPR22,
YEAR = "2022",
PAGES = "18041-18050",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240799"}
@inproceedings{bb245891,
AUTHOR = "Khandelwal, A. and Weihs, L. and Mottaghi, R. and Kembhavi, A.",
TITLE = "Simple but Effective: CLIP Embeddings for Embodied AI",
BOOKTITLE = CVPR22,
YEAR = "2022",
PAGES = "14809-14818",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240800"}
@inproceedings{bb245892,
AUTHOR = "Ma, H.Y. and Zhao, H. and Lin, Z. and Kale, A. and Wang, Z.Y. and Yu, T. and Gu, J.X. and Choudhary, S. and Xie, X.H.",
TITLE = "EI-CLIP: Entity-aware Interventional Contrastive Learning for
E-commerce Cross-modal Retrieval",
BOOKTITLE = CVPR22,
YEAR = "2022",
PAGES = "18030-18040",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240801"}
@inproceedings{bb245893,
AUTHOR = "Barraco, M. and Cornia, M. and Cascianelli, S. and Baraldi, L. and Cucchiara, R.",
TITLE = "The Unreasonable Effectiveness of CLIP Features for Image Captioning:
An Experimental Analysis",
BOOKTITLE = MULA22,
YEAR = "2022",
PAGES = "4661-4669",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240802"}
@inproceedings{bb245894,
AUTHOR = "Tevet, G. and Gordon, B. and Hertz, A. and Bermano, A.H. and Cohen Or, D.",
TITLE = "MotionCLIP: Exposing Human Motion Generation to CLIP Space",
BOOKTITLE = ECCV22,
YEAR = "2022",
PAGES = "XXII:358-374",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240803"}
@inproceedings{bb245895,
AUTHOR = "Materzynska, J. and Torralba, A. and Bau, D.",
TITLE = "Disentangling visual and written concepts in CLIP",
BOOKTITLE = CVPR22,
YEAR = "2022",
PAGES = "16389-16398",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240804"}
@inproceedings{bb245896,
AUTHOR = "Li, M. and Xu, R. and Wang, S. and Zhou, L. and Lin, X.D. and Zhu, C.G. and Zeng, M. and Ji, H. and Chang, S.F.",
TITLE = "CLIP-Event: Connecting Text and Images with Event Structures",
BOOKTITLE = CVPR22,
YEAR = "2022",
PAGES = "16399-16408",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240805"}
@inproceedings{bb245897,
AUTHOR = "Zhong, Y. and Yang, J.W. and Zhang, P.C. and Li, C.Y. and Codella, N. and Li, L.H. and Zhou, L. and Dai, X. and Yuan, L. and Li, Y. and Gao, J.F.",
TITLE = "RegionCLIP: Region-based Language-Image Pretraining",
BOOKTITLE = CVPR22,
YEAR = "2022",
PAGES = "16772-16782",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240806"}
@inproceedings{bb245898,
AUTHOR = "Patashnik, O. and Wu, Z.Z. and Shechtman, E. and Cohen Or, D. and Lischinski, D.",
TITLE = "StyleCLIP: Text-Driven Manipulation of StyleGAN Imagery",
BOOKTITLE = ICCV21,
YEAR = "2021",
PAGES = "2065-2074",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803clip3.html#TT240807"}
@article{bb245899,
AUTHOR = "Su, H.H. and Chen, T.W. and Kao, C.C. and Hsu, W.H. and Chien, S.Y.",
TITLE = "Preference-Aware View Recommendation System for Scenic Photos Based on
Bag-of-Aesthetics-Preserving Features",
JOURNAL = MultMed,
VOLUME = "14",
YEAR = "2012",
NUMBER = "3",
PAGES = "833-843",
BIBSOURCE = "http://www.visionbib.com/bibliography/applicat803we1.html#TT240808"}
Last update:Jul 24, 2026 at 15:25:55