@article{bb160300,
        AUTHOR = "Liao, Y. and Gao, Y.S. and Zhang, W.C.",
        TITLE = "Dynamic accumulated attention map for interpreting evolution of
decision-making in vision transformer",
        JOURNAL = PR,
        VOLUME = "165",
        YEAR = "2025",
        PAGES = "111607",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156164"}

@article{bb160301,
        AUTHOR = "Shi, Y.L. and Sun, M.W. and Wang, Y.S. and Ma, J.H. and Chen, Z.Q.",
        TITLE = "EViT: An Eagle Vision Transformer With Bi-Fovea Self-Attention",
        JOURNAL = Cyber,
        VOLUME = "55",
        YEAR = "2025",
        NUMBER = "3",
        MONTH = "March",
        PAGES = "1288-1300",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156165"}

@article{bb160302,
        AUTHOR = "Long, W. and Chen, Z.Y. and Li, W.T. and Zhang, Y.J. and Yao, H. and Peng, J.X. and Cui, Z.W.",
        TITLE = "Leveraging negative correlation for Full-Range Self-Attention in
Vision Transformers",
        JOURNAL = PR,
        VOLUME = "169",
        YEAR = "2026",
        PAGES = "111899",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156166"}

@article{bb160303,
        AUTHOR = "Shan, J. and Wang, J.X. and Zhao, L.F. and Cai, L. and Zhang, H.Y. and Liritzis, I.",
        TITLE = "AnchorFormer: Differentiable anchor attention for efficient vision
transformer",
        JOURNAL = PRL,
        VOLUME = "197",
        YEAR = "2025",
        PAGES = "124-131",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156167"}

@article{bb160304,
        AUTHOR = "Bae, J. and Kim, S. and Cho, M. and Kim, H.Y.",
        TITLE = "MVFormer: Diversifying feature normalization and token mixing for
efficient vision transformers",
        JOURNAL = PRL,
        VOLUME = "197",
        YEAR = "2025",
        PAGES = "72-80",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156168"}

@article{bb160305,
        AUTHOR = "Li, Y. and Jiao, L.C. and Liu, X. and Liu, F. and Li, L.L. and Chen, P.",
        TITLE = "Semantic-Aware Wavelet Transformer for Pyramid Learning Object
Detection",
        JOURNAL = MultMed,
        VOLUME = "27",
        YEAR = "2025",
        PAGES = "8016-8028",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156169"}

@article{bb160306,
        AUTHOR = "Liu, Z. and Rao, Y.M. and Zhao, W.L. and Zhou, J. and Lu, J.W.",
        TITLE = "Efficient High-Order Spatial Interactions for Visual Perception",
        JOURNAL = PAMI,
        VOLUME = "48",
        YEAR = "2026",
        NUMBER = "1",
        MONTH = "January",
        PAGES = "33-46",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156170"}

@inproceedings{bb160307,
        AUTHOR = "Rao, Y.M. and Zhao, W.L. and Zhou, J. and Lu, J.W.",
        TITLE = "AMixer:
Adaptive Weight Mixing for Self-Attention Free Vision Transformers",
        BOOKTITLE = ECCV22,
        YEAR = "2022",
        PAGES = "XXI:50-67",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156171"}

@article{bb160308,
        AUTHOR = "Guo, H.L. and Lv, W.J. and Shen, Z. and Wang, D.H. and Zhang, Y.",
        TITLE = "CPFormer-Net: Correspondence Pruning Transformer With Structured
Context Aggregation",
        JOURNAL = SPLetters,
        VOLUME = "33",
        YEAR = "2026",
        PAGES = "111-115",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156172"}

@article{bb160309,
        AUTHOR = "Hang, J.F. and Yang, X.Q.",
        TITLE = "Enhancing local attention with global information interaction via
progressive cluster propagation",
        JOURNAL = PR,
        VOLUME = "172",
        YEAR = "2026",
        PAGES = "112713",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156173"}

@article{bb160310,
        AUTHOR = "Lin, S.H. and Lyu, P.M. and Liu, D.R. and Li, Z.H. and Wang, W.G. and Chang, X.J. and Zheng, Y.H.",
        TITLE = "Entropy-Guided Condensing for Vision Transformer",
        JOURNAL = IJCV,
        VOLUME = "134",
        YEAR = "2026",
        NUMBER = "1",
        MONTH = "January",
        PAGES = "86",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156174"}

@article{bb160311,
        AUTHOR = "Wu, C. and Che, M.L. and Yan, H.",
        TITLE = "The CUR Decomposition of Self-Attention Matrices in Vision
Transformers",
        JOURNAL = PAMI,
        VOLUME = "48",
        YEAR = "2026",
        NUMBER = "4",
        MONTH = "April",
        PAGES = "4792-4809",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156175"}

@article{bb160312,
        AUTHOR = "Li, Y. and Wu, X. and Wang, J.C. and Bo, Y.M. and Ni, F. and Jiang, C.H.",
        TITLE = "Differential attention vision transformer with adaptive spatial
feature conditioning for remote sensing scene classification",
        JOURNAL = PR,
        VOLUME = "178",
        YEAR = "2026",
        PAGES = "113461",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156176"}

@article{bb160313,
        AUTHOR = "Liu, Y.H. and Wen, Y. and Yang, L.Z. and He, L.H. and Zhou, M.",
        TITLE = "A General Framework for Efficient Medical Image Analysis via Shared
Attention Vision Transformer",
        JOURNAL = MedImg,
        VOLUME = "45",
        YEAR = "2026",
        NUMBER = "5",
        MONTH = "May",
        PAGES = "2001-2014",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156177"}

@article{bb160314,
        AUTHOR = "Zhou, S. and Liu, M. and Zhou, J. and Zheng, R.H.",
        TITLE = "Enhancing Vision Transformer With Shift Expansion Linear Attention
for Image Classification and Object Tracking",
        JOURNAL = CirSysVideo,
        VOLUME = "36",
        YEAR = "2026",
        NUMBER = "6",
        MONTH = "June",
        PAGES = "9042-9056",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156178"}

@article{bb160315,
        AUTHOR = "Waseem, A. and Ruiu, P. and Nixon, S. and Lagorio, A. and Tistarelli, M.",
        TITLE = "Self-attention as the backbone: A survey on Vision Transformers",
        JOURNAL = CVIU,
        VOLUME = "269",
        YEAR = "2026",
        PAGES = "104791",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156179"}

@article{bb160316,
        AUTHOR = "Qi, N. and Zhao, P. and Wang, G.Q. and Zhao, C. and Yang, S.",
        TITLE = "Independent Block-Wise Attribution for Vision Transformer
Interpretability Through Semantic Relevance",
        JOURNAL = MultMed,
        VOLUME = "28",
        YEAR = "2026",
        PAGES = "5303-5314",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156180"}

@article{bb160317,
        AUTHOR = "Guan, S.Q. and Liang, W.X. and Gao, Y.L. and Zong, L.L. and Liu, X. and Zhang, X.C.",
        TITLE = "Hybrid linear attention: A vision transformer integrating selective
sampling softmax and multi-feature fusion enhancement",
        JOURNAL = PR,
        VOLUME = "180",
        YEAR = "2026",
        PAGES = "114037",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156181"}

@inproceedings{bb160318,
        AUTHOR = "Jo, S. and Jang, G. and Park, H.",
        TITLE = "GMAR: Gradient-Driven Multi-Head Attention Rollout for Vision
Transformer Interpretability",
        BOOKTITLE = ICIP25,
        YEAR = "2025",
        PAGES = "582-587",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156182"}

@inproceedings{bb160319,
        AUTHOR = "Savathrakis, G. and Argyros, A.",
        TITLE = "Enact: Entropy-Based Clustering of Attention Input for Reducing the
Computational Needs of Object Detection Transformers",
        BOOKTITLE = ICIP25,
        YEAR = "2025",
        PAGES = "295-300",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156183"}

@inproceedings{bb160320,
        AUTHOR = "Fan, Q.H. and Huang, H.B. and He, R.",
        TITLE = "Breaking the Low-Rank Dilemma of Linear Attention",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "25271-25280",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156184"}

@inproceedings{bb160321,
        AUTHOR = "Miao, Z.C. and Chen, W. and Qiu, Q.",
        TITLE = "Coeff-Tuning: A Graph Filter Subspace View for Tuning Attention-Based
Large Models",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "20146-20146",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156185"}

@inproceedings{bb160322,
        AUTHOR = "Sun, Y.W. and Ochiai, H. and Wu, Z.R. and Lin, S. and Kanai, R.",
        TITLE = "Associative Transformer",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "4518-4527",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156186"}

@inproceedings{bb160323,
        AUTHOR = "Chen, L.Y. and Meyer, G.P. and Zhang, Z. and Wolff, E.M. and Vernaza, P.",
        TITLE = "Flash3D: Super-scaling Point Transformers through Joint
Hardware-Geometry Locality",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "6595-6604",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156187"}

@inproceedings{bb160324,
        AUTHOR = "Zhang, W. and Zhang, B.P. and Teng, Z. and Luo, W.X. and Zou, J. and Fan, J.P.",
        TITLE = "Less Attention is More: Prompt Transformer for Generalized Category
Discovery",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "30322-30331",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156188"}

@inproceedings{bb160325,
        AUTHOR = "Zhu, J.C. and Chen, X.L. and He, K. and LeCun, Y. and Liu, Z.",
        TITLE = "Transformers without Normalization",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "14901-14911",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156189"}

@inproceedings{bb160326,
        AUTHOR = "Peng, Z.L. and Huang, Y. and Xu, Z.Q. and Tang, F.L. and Hu, M. and Yang, X.K. and Shen, W.",
        TITLE = "Star with Bilinear Mapping",
        BOOKTITLE = CVPR25,
        YEAR = "2025",
        PAGES = "25292-25302",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156190"}

@inproceedings{bb160327,
        AUTHOR = "Nottebaum, M. and Dunnhofer, M. and Micheloni, C.",
        TITLE = "LowFormer: Hardware Efficient Design for Convolutional Transformer
Backbones",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "7008-7018",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156191"}

@inproceedings{bb160328,
        AUTHOR = "Chowdhury, A.R. and Diddigi, R.B. and Prabuchandran, K.J. and Tripathi, A.M.",
        TITLE = "Bandit-based Attention Mechanism in Vision Transformers",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "9597-9606",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156192"}

@inproceedings{bb160329,
        AUTHOR = "Alam, Q.M. and Tarchoun, B. and Alouani, I. and Abu Ghazaleh, N.",
        TITLE = "Adversarial Attention Deficit: Fooling Deformable Vision Transformers
with Collaborative Adversarial Patches",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "7123-7132",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156193"}

@inproceedings{bb160330,
        AUTHOR = "Ren, S. and Zhou, D. and He, S.F. and Feng, J.S. and Wang, X.C.",
        TITLE = "Shunted Self-Attention via Multi-Scale Token Aggregation",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "10843-10852",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156194"}

@inproceedings{bb160331,
        AUTHOR = "Qiang, Y. and Li, C.Y. and Khanduri, P. and Zhu, D.X.",
        TITLE = "Fairness-aware Vision Transformer via Debiased Self-attention",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "XXXVII: 358-376",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156195"}

@inproceedings{bb160332,
        AUTHOR = "Gong, H.H. and Dong, M.J. and Ma, S.Q. and Camtepe, S. and Nepal, S. and Xu, C.",
        TITLE = "Random Entangled Tokens for Adversarially Robust Vision Transformer",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "24554-24563",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156196"}

@inproceedings{bb160333,
        AUTHOR = "Lee, S. and Choi, J. and Kim, H.W.J.",
        TITLE = "Multi-Criteria Token Fusion with One-Step-Ahead Attention for
Efficient Vision Transformers",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "15741-15750",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156197"}

@inproceedings{bb160334,
        AUTHOR = "Zhang, S.X. and Liu, H.P. and Lin, S. and He, K.",
        TITLE = "You Only Need Less Attention at Each Stage in Vision Transformers",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "6057-6066",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156198"}

@inproceedings{bb160335,
        AUTHOR = "Li, L. and Wei, Z. and Dong, P. and Luo, W.H. and Xue, W. and Liu, Q.F. and Guo, Y.",
        TITLE = "Attnzero: Efficient Attention Discovery for Vision Transformers",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "V: 20-37",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156199"}

@inproceedings{bb160336,
        AUTHOR = "Bao Long, N.H. and Zhang, C.Y. and Shi, Y.Z. and Hirakawa, T. and Yamashita, T. and Matsui, T. and Fujiyoshi, H.",
        TITLE = "Debiformer: Vision Transformer with Deformable Agent Bi-level Routing
Attention",
        BOOKTITLE = ACCV24,
        YEAR = "2024",
        PAGES = "X: 445-462",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156200"}

@inproceedings{bb160337,
        AUTHOR = "Yang, X. and Yuan, L.Z. and Wilber, K. and Sharma, A. and Gu, X.Y. and Qiao, S.Y. and Debats, S. and Wang, H.S. and Adam, H. and Sirotenko, M. and Chen, L.C.",
        TITLE = "PolyMaX: General Dense Prediction with Mask Transformer",
        BOOKTITLE = WACV24,
        YEAR = "2024",
        PAGES = "1039-1050",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156201"}

@inproceedings{bb160338,
        AUTHOR = "Nie, X.S. and Chen, X. and Jin, H.Y. and Zhu, Z.H. and Yan, Y.F. and Qi, D.L.",
        TITLE = "Triplet Attention Transformer for Spatiotemporal Predictive Learning",
        BOOKTITLE = WACV24,
        YEAR = "2024",
        PAGES = "7021-7030",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156202"}

@inproceedings{bb160339,
        AUTHOR = "Cai, H. and Li, J. and Hu, M. and Gan, C. and Han, S.",
        TITLE = "EfficientViT: Lightweight Multi-Scale Attention for High-Resolution
Dense Prediction",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "17256-17267",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156203"}

@inproceedings{bb160340,
        AUTHOR = "Ryu, J.B. and Han, D.Y. and Lim, J.W.",
        TITLE = "Gramian Attention Heads are Strong yet Efficient Vision Learners",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "5818-5828",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156204"}

@inproceedings{bb160341,
        AUTHOR = "Xu, R.H. and Zhang, H. and Hu, W.Z. and Zhang, S.L. and Wang, X.Y.",
        TITLE = "ParCNetV2: Oversized Kernel with Enhanced Attention*",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "5729-5739",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156205"}

@inproceedings{bb160342,
        AUTHOR = "Zhao, B.Y. and Yu, Z. and Lan, S.Y. and Cheng, Y.T. and Anandkumar, A. and Lao, Y.J. and Alvarez, J.M.",
        TITLE = "Fully Attentional Networks with Self-emerging Token Labeling",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "5562-5572",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156206"}

@inproceedings{bb160343,
        AUTHOR = "Guo, Y. and Stutz, D. and Schiele, B.",
        TITLE = "Robustifying Token Attention for Vision Transformers",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "17511-17522",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156207"}

@inproceedings{bb160344,
        AUTHOR = "Zhao, Y.P. and Tang, H.D. and Jiang, Y.Y. and A, Y. and Wu, Q. and Wang, J.",
        TITLE = "Parameter-Efficient Vision Transformer with Linear Attention",
        BOOKTITLE = ICIP23,
        YEAR = "2023",
        PAGES = "1275-1279",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156208"}

@inproceedings{bb160345,
        AUTHOR = "Shi, L. and Huang, H.D. and Song, B. and Tan, M. and Zhao, W.Z. and Xia, T. and Ren, P.J.",
        TITLE = "TAQ: Top-K Attention-Aware Quantization for Vision Transformers",
        BOOKTITLE = ICIP23,
        YEAR = "2023",
        PAGES = "1750-1754",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156209"}

@inproceedings{bb160346,
        AUTHOR = "Baili, N. and Frigui, H.",
        TITLE = "ADA-VIT: Attention-Guided Data Augmentation for Vision Transformers",
        BOOKTITLE = ICIP23,
        YEAR = "2023",
        PAGES = "385-389",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156210"}

@inproceedings{bb160347,
        AUTHOR = "Ding, M.Y. and Shen, Y.K. and Fan, L.J. and Chen, Z.F. and Chen, Z. and Luo, P. and Tenenbaum, J. and Gan, C.",
        TITLE = "Visual Dependency Transformers:
Dependency Tree Emerges from Reversed Attention",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "14528-14539",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156211"}

@inproceedings{bb160348,
        AUTHOR = "Song, J.C. and Mou, C. and Wang, S.Q. and Ma, S.W. and Zhang, J.",
        TITLE = "Optimization-Inspired Cross-Attention Transformer for Compressive
Sensing",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "6174-6184",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156212"}

@inproceedings{bb160349,
        AUTHOR = "Hassani, A. and Walton, S. and Li, J.C. and Li, S. and Shi, H.",
        TITLE = "Neighborhood Attention Transformer",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "6185-6194",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156213"}

@inproceedings{bb160350,
        AUTHOR = "Liu, Z.J. and Yang, X.Y. and Tang, H.T. and Yang, S. and Han, S.",
        TITLE = "FlatFormer: Flattened Window Attention for Efficient Point Cloud
Transformer",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "1200-1211",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156214"}

@inproceedings{bb160351,
        AUTHOR = "Pan, X. and Ye, T.Z. and Xia, Z.F. and Song, S. and Huang, G.",
        TITLE = "Slide-Transformer: Hierarchical Vision Transformer with Local
Self-Attention",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "2082-2091",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156215"}

@inproceedings{bb160352,
        AUTHOR = "Zhu, L. and Wang, X.J. and Ke, Z.H. and Zhang, W. and Lau, R.",
        TITLE = "BiFormer: Vision Transformer with Bi-Level Routing Attention",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "10323-10333",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156216"}

@inproceedings{bb160353,
        AUTHOR = "Long, S. and Zhao, Z. and Pi, J. and Wang, S.S. and Wang, J.D.",
        TITLE = "Beyond Attentive Tokens: Incorporating Token Importance and Diversity
for Efficient Vision Transformers",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "10334-10343",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156217"}

@inproceedings{bb160354,
        AUTHOR = "Liu, X.Y. and Peng, H. and Zheng, N.X. and Yang, Y.Q. and Hu, H. and Yuan, Y.X.",
        TITLE = "EfficientViT: Memory Efficient Vision Transformer with Cascaded Group
Attention",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "14420-14430",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156218"}

@inproceedings{bb160355,
        AUTHOR = "You, H.R. and Xiong, Y. and Dai, X.L. and Wu, B. and Zhang, P.Z. and Fan, H.Q. and Vajda, P. and Lin, Y.Y.C.",
        TITLE = "Castling-ViT: Compressing Self-Attention via Switching Towards
Linear-Angular Attention at Vision Transformer Inference",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "14431-14442",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156219"}

@inproceedings{bb160356,
        AUTHOR = "Grainger, R. and Paniagua, T. and Song, X. and Cuntoor, N. and Lee, M.W. and Wu, T.F.",
        TITLE = "PaCa-ViT: Learning Patch-to-Cluster Attention in Vision Transformers",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "18568-18578",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156220"}

@inproceedings{bb160357,
        AUTHOR = "Wei, C. and Duke, B. and Jiang, R. and Aarabi, P. and Taylor, G.W. and Shkurti, F.",
        TITLE = "Sparsifiner: Learning Sparse Instance-Dependent Attention for
Efficient Vision Transformers",
        BOOKTITLE = CVPR23,
        YEAR = "2023",
        PAGES = "22680-22689",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156221"}

@inproceedings{bb160358,
        AUTHOR = "Bhattacharyya, M. and Chattopadhyay, S. and Nag, S.",
        TITLE = "DeCAtt: Efficient Vision Transformers with Decorrelated Attention
Heads",
        BOOKTITLE = ECV23,
        YEAR = "2023",
        PAGES = "4695-4699",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156222"}

@inproceedings{bb160359,
        AUTHOR = "Zhang, Y. and Chen, D. and Kundu, S. and Li, C.H. and Beerel, P.A.",
        TITLE = "SAL-ViT: Towards Latency Efficient Private Inference on ViT using
Selective Attention Search with a Learnable Softmax Approximation",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "5093-5102",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156223"}

@inproceedings{bb160360,
        AUTHOR = "Yeganeh, Y. and Farshad, A. and Weinberger, P. and Ahmadi, S.A. and Adeli, E. and Navab, N.",
        TITLE = "Transformers Pay Attention to Convolutions Leveraging Emerging
Properties of ViTs by Dual Attention-Image Network",
        BOOKTITLE = CVAMD23,
        YEAR = "2023",
        PAGES = "2296-2307",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156224"}

@inproceedings{bb160361,
        AUTHOR = "Zheng, J.H. and Yang, L.Q. and Li, Y.Y. and Yang, K. and Wang, Z.Y. and Zhou, J.",
        TITLE = "Lightweight Vision Transformer with Spatial and Channel Enhanced
Self-Attention",
        BOOKTITLE = REDLCV23,
        YEAR = "2023",
        PAGES = "1484-1488",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156225"}

@inproceedings{bb160362,
        AUTHOR = "Hyeon Woo, N. and Yu Ji, K. and Heo, B. and Han, D.Y. and Oh, S.J. and Oh, T.H.",
        TITLE = "Scratching Visual Transformer's Back with Uniform Attention",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "5784-5795",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156226"}

@inproceedings{bb160363,
        AUTHOR = "Zhang, H.K. and Hu, W.Z. and Wang, X.Y.",
        TITLE = "Fcaformer: Forward Cross Attention in Hybrid Vision Transformer",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "6037-6046",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156227"}

@inproceedings{bb160364,
        AUTHOR = "Zeng, W.X. and Li, M. and Xiong, W.J. and Tong, T. and Lu, W.J. and Tan, J. and Wang, R.S. and Huang, R.",
        TITLE = "MPCViT: Searching for Accurate and Efficient MPC-Friendly Vision
Transformer with Heterogeneous Attention",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "5029-5040",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156228"}

@inproceedings{bb160365,
        AUTHOR = "Psomas, B. and Kakogeorgiou, I. and Karantzalos, K. and Avrithis, Y.",
        TITLE = "Keep It SimPool:Who Said Supervised Transformers Suffer from
Attention Deficit?",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "5327-5337",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156229"}

@inproceedings{bb160366,
        AUTHOR = "Han, D.C. and Pan, X. and Han, Y.Z. and Song, S. and Huang, G.",
        TITLE = "FLatten Transformer: Vision Transformer using Focused Linear
Attention",
        BOOKTITLE = ICCV23,
        YEAR = "2023",
        PAGES = "5938-5948",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156230"}

@inproceedings{bb160367,
        AUTHOR = "Tatsunami, Y. and Taki, M.",
        TITLE = "RaftMLP: How Much Can Be Done Without Attention and with Less Spatial
Locality?",
        BOOKTITLE = ACCV22,
        YEAR = "2022",
        PAGES = "VI:459-475",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156231"}

@inproceedings{bb160368,
        AUTHOR = "Bolya, D. and Fu, C.Y. and Dai, X.L. and Zhang, P.Z. and Hoffman, J.",
        TITLE = "Hydra Attention: Efficient Attention with Many Heads",
        BOOKTITLE = CADK22,
        YEAR = "2022",
        PAGES = "35-49",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156232"}

@inproceedings{bb160369,
        AUTHOR = "Chen, X.Y. and Hu, Q.H. and Li, K. and Zhong, C. and Wang, G.H.",
        TITLE = "Accumulated Trivial Attention Matters in Vision Transformers on Small
Datasets",
        BOOKTITLE = WACV23,
        YEAR = "2023",
        PAGES = "3973-3981",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156233"}

@inproceedings{bb160370,
        AUTHOR = "Lan, H. and Wang, X. and Shen, H. and Liang, P.D. and Wei, X.",
        TITLE = "Couplformer: Rethinking Vision Transformer with Coupling Attention",
        BOOKTITLE = WACV23,
        YEAR = "2023",
        PAGES = "6464-6473",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156234"}

@inproceedings{bb160371,
        AUTHOR = "Debnath, B. and Po, O. and Chowdhury, F.A. and Chakradhar, S.",
        TITLE = "Cosine Similarity based Few-Shot Video Classifier with
Attention-based Aggregation",
        BOOKTITLE = "ICPR22",
        YEAR = "2022",
        PAGES = "1273-1279",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156235"}

@inproceedings{bb160372,
        AUTHOR = "Mari, C.R. and Gonzalez, D.V. and Bou Balust, E.",
        TITLE = "Multi-Scale Transformer-Based Feature Combination for Image Retrieval",
        BOOKTITLE = ICIP22,
        YEAR = "2022",
        PAGES = "3166-3170",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156236"}

@inproceedings{bb160373,
        AUTHOR = "Furukawa, R. and Hotta, K.",
        TITLE = "Local Embedding for Axial Attention",
        BOOKTITLE = ICIP22,
        YEAR = "2022",
        PAGES = "2586-2590",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156237"}

@inproceedings{bb160374,
        AUTHOR = "Ding, M.Y. and Xiao, B. and Codella, N. and Luo, P. and Wang, J.D. and Yuan, L.",
        TITLE = "DaViT: Dual Attention Vision Transformers",
        BOOKTITLE = ECCV22,
        YEAR = "2022",
        PAGES = "XXIV:74-92",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156238"}

@inproceedings{bb160375,
        AUTHOR = "Wang, P.C. and Wang, X. and Wang, F. and Lin, M. and Chang, S.N. and Li, H. and Jin, R.",
        TITLE = "KVT: k-NN Attention for Boosting Vision Transformers",
        BOOKTITLE = ECCV22,
        YEAR = "2022",
        PAGES = "XXIV:285-302",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156239"}

@inproceedings{bb160376,
        AUTHOR = "Li, A. and Jiao, J.C. and Li, N. and Qi, W.J. and Xu, W. and Pang, M.",
        TITLE = "Conmw Transformer: A General Vision Transformer Backbone With
Merged-Window Attention",
        BOOKTITLE = ICIP22,
        YEAR = "2022",
        PAGES = "1551-1555",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156240"}

@inproceedings{bb160377,
        AUTHOR = "Zhang, Q.M. and Xu, Y.F. and Zhang, J. and Tao, D.C.",
        TITLE = "VSA: Learning Varied-Size Window Attention in Vision Transformers",
        BOOKTITLE = ECCV22,
        YEAR = "2022",
        PAGES = "XXV:466-483",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156241"}

@inproceedings{bb160378,
        AUTHOR = "Mallick, R. and Benois Pineau, J. and Zemmari, A.",
        TITLE = "I Saw: A Self-Attention Weighted Method for Explanation of Visual
Transformers",
        BOOKTITLE = ICIP22,
        YEAR = "2022",
        PAGES = "3271-3275",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156242"}

@inproceedings{bb160379,
        AUTHOR = "Song, Z.K. and Yu, J.Q. and Chen, Y.P.P. and Yang, W.",
        TITLE = "Transformer Tracking with Cyclic Shifting Window Attention",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "8781-8790",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156243"}

@inproceedings{bb160380,
        AUTHOR = "Yang, C.L. and Wang, Y.L. and Zhang, J.M. and Zhang, H. and Wei, Z.J. and Lin, Z. and Yuille, A.L.",
        TITLE = "Lite Vision Transformer with Enhanced Self-Attention",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "11988-11998",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156244"}

@inproceedings{bb160381,
        AUTHOR = "Xia, Z.F. and Pan, X. and Song, S. and Li, L.E. and Huang, G.",
        TITLE = "Vision Transformer with Deformable Attention",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "4784-4793",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156245"}

@inproceedings{bb160382,
        AUTHOR = "Yu, T. and Khalitov, R. and Cheng, L. and Yang, Z.R.",
        TITLE = "Paramixer: Parameterizing Mixing Links in Sparse Factors Works Better
than Dot-Product Self-Attention",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "681-690",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156246"}

@inproceedings{bb160383,
        AUTHOR = "Cheng, B. and Misra, I. and Schwing, A.G. and Kirillov, A. and Girdhar, R.",
        TITLE = "Masked-attention Mask Transformer for Universal Image Segmentation",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "1280-1289",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156247"}

@inproceedings{bb160384,
        AUTHOR = "Rangrej, S.B. and Srinidhi, C.L. and Clark, J.J.",
        TITLE = "Consistency driven Sequential Transformers Attention Model for
Partially Observable Scenes",
        BOOKTITLE = CVPR22,
        YEAR = "2022",
        PAGES = "2508-2517",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156248"}

@inproceedings{bb160385,
        AUTHOR = "Chen, C.F.R. and Fan, Q.F. and Panda, R.",
        TITLE = "CrossViT: Cross-Attention Multi-Scale Vision Transformer for Image
Classification",
        BOOKTITLE = ICCV21,
        YEAR = "2021",
        PAGES = "347-356",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156249"}

@inproceedings{bb160386,
        AUTHOR = "Chefer, H. and Gur, S. and Wolf, L.B.",
        TITLE = "Generic Attention-model Explainability for Interpreting Bi-Modal and
Encoder-Decoder Transformers",
        BOOKTITLE = ICCV21,
        YEAR = "2021",
        PAGES = "387-396",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156250"}

@inproceedings{bb160387,
        AUTHOR = "Xu, W.J. and Xu, Y.F. and Chang, T. and Tu, Z.W.",
        TITLE = "Co-Scale Conv-Attentional Image Transformers",
        BOOKTITLE = ICCV21,
        YEAR = "2021",
        PAGES = "9961-9970",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156251"}

@inproceedings{bb160388,
        AUTHOR = "Yang, G.L. and Tang, H. and Ding, M.L. and Sebe, N. and Ricci, E.",
        TITLE = "Transformer-Based Attention Networks for Continuous Pixel-Wise
Prediction",
        BOOKTITLE = ICCV21,
        YEAR = "2021",
        PAGES = "16249-16259",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156252"}

@inproceedings{bb160389,
        AUTHOR = "Kim, K. and Wu, B.C. and Dai, X.L. and Zhang, P.Z. and Yan, Z.C. and Vajda, P. and Kim, S.",
        TITLE = "Rethinking the Self-Attention in Vision Transformers",
        BOOKTITLE = ECV21,
        YEAR = "2021",
        PAGES = "3065-3069",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651atvit4.html#TT156253"}

@article{bb160390,
        AUTHOR = "Yang, J.H. and Li, X.Y. and Zheng, M. and Wang, Z.H. and Zhu, Y.Q. and Guo, X.Q. and Yuan, Y.C. and Chai, Z. and Jiang, S.Q.",
        TITLE = "MemBridge: Video-Language Pre-Training With Memory-Augmented
Inter-Modality Bridge",
        JOURNAL = IP,
        VOLUME = "32",
        YEAR = "2023",
        PAGES = "4073-4087",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651vidt3.html#TT156254"}

@article{bb160391,
        AUTHOR = "Selva, J. and Johansen, A.S. and Escalera, S. and Nasrollahi, K. and Moeslund, T.B. and Clapes, A.",
        TITLE = "Video Transformers: A Survey",
        JOURNAL = PAMI,
        VOLUME = "45",
        YEAR = "2023",
        NUMBER = "11",
        MONTH = "November",
        PAGES = "12922-12943",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651vidt3.html#TT156255"}

@article{bb160392,
        AUTHOR = "Zhang, Z.C. and Chen, Z.D. and Wang, Y.X. and Luo, X. and Xu, X.S.",
        TITLE = "A vision transformer for fine-grained classification by reducing
noise and enhancing discriminative information",
        JOURNAL = PR,
        VOLUME = "145",
        YEAR = "2024",
        PAGES = "109979",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651vidt3.html#TT156256"}

@article{bb160393,
        AUTHOR = "Xian, K. and Peng, J. and Cao, Z.G. and Zhang, J.M. and Lin, G.S.",
        TITLE = "ViTA: Video Transformer Adaptor for Robust Video Depth Estimation",
        JOURNAL = MultMed,
        VOLUME = "26",
        YEAR = "2024",
        PAGES = "3302-3316",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651vidt3.html#TT156257"}

@article{bb160394,
        AUTHOR = "Zhang, J.S. and Gu, L.F. and Lai, Y.K. and Wang, X.Y. and Li, K.",
        TITLE = "Toward Grouping in Large Scenes With Occlusion-Aware Spatio-Temporal
Transformers",
        JOURNAL = CirSysVideo,
        VOLUME = "34",
        YEAR = "2024",
        NUMBER = "5",
        MONTH = "May",
        PAGES = "3919-3929",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651vidt3.html#TT156258"}

@inproceedings{bb160395,
        AUTHOR = "Goyal, R. and Fan, W.C. and Siam, M. and Sigal, L.",
        TITLE = "TAM-VT: Transformation-Aware Multi-Scale Video Transformer for
Segmentation and Tracking",
        BOOKTITLE = WACV25,
        YEAR = "2025",
        PAGES = "8336-8345",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651vidt3.html#TT156259"}

@inproceedings{bb160396,
        AUTHOR = "Wu, R. and Zhou, F.X. and Yin, Z.W. and Liu, K.J.",
        TITLE = "Aligning Neuronal Coding of Dynamic Visual Scenes with Foundation
Vision Models",
        BOOKTITLE = ECCV24,
        YEAR = "2024",
        PAGES = "LXXXVIII: 238-254",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651vidt3.html#TT156260"}

@inproceedings{bb160397,
        AUTHOR = "Lu, Y.W. and Liu, D.F. and Wang, Q.F. and Han, C. and Cui, Y.M. and Cao, Z.W. and Zhang, X.L. and Chen, Y.J.V. and Fan, H.",
        TITLE = "ProMotion: Prototypes as Motion Learners",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "28109-28119",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651vidt3.html#TT156261"}

@inproceedings{bb160398,
        AUTHOR = "Choi, J. and Lee, S. and Chu, J.W. and Choi, M. and Kim, H.W.J.",
        TITLE = "vid-TLDR: Training Free Token merging for Light-Weight Video
Transformer",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "18771-18781",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651vidt3.html#TT156262"}

@inproceedings{bb160399,
        AUTHOR = "Kowal, M. and Dave, A. and Ambrus, R. and Gaidon, A. and Derpanis, K.G. and Tokmakov, P.",
        TITLE = "Understanding Video Transformers via Universal Concept Discovery",
        BOOKTITLE = CVPR24,
        YEAR = "2024",
        PAGES = "10946-10956",
        BIBSOURCE = "http://www.visionbib.com/bibliography/pattern651vidt3.html#TT156263"}

Last update:Aug 19, 2026 at 13:26:35