@inproceedings{7e2e63c65329456d872da0a3b4960441,
title = "Gate-ViT: gated vision transformer for Fine-Grained Visual Classification",
abstract = "Fine-Grained Visual Classification (FGVC) aims to distinguish images with subtle differences and limited inter-class variation. This study addresses the lack of understanding of dataset intricacies by analyzing the CUB-200-2011 (CUB) dataset. We identify four ultra-fine-grained subsets that significantly impact accuracy. While local visual cues can often distinguish samples, some fine-grained cases require global context for accurate classification. To address this, we propose a novel LSTM Decision Module (LSTM-DM) that combines local and global information flexibly. Additionally, a Region Selection Module (RSM) selects discriminative regions of fine-grained samples. These modules are integrated into the ViT architecture, enhancing its performance on FGVC tasks. A contrastive loss further improves feature representation by increasing the distance between confusing classes. Extensive evaluations show that our Gate-ViT outperforms existing state-of-the-art methods on four benchmark datasets.",
keywords = "Fine-Grained Visual Classification, Vision Transformer, LSTM",
author = "Xiaowei Lu and Kanqi Wang and Peiyu Wang and Qin Zhang and Yang Zhao and Gang Liu and Xiaohan Yu",
year = "2025",
doi = "10.1007/978-981-96-8180-8\_37",
language = "English",
isbn = "9789819681792",
series = "Lecture Notes in Computer Science",
publisher = "Springer, Springer Nature",
pages = "468--479",
editor = "Xintao Wu and Myra Spiliopoulou and Can Wang and Vipin Kumar and Longbing Cao and Yanqiu Wu and Yu Yao and Zhangkai Wu",
booktitle = "Advances in Knowledge Discovery and Data Mining",
address = "United States",
note = "29th Pacific-Asia Conference on Knowledge Discovery and Data Mining, PAKDD 2025 ; Conference date: 10-06-2025 Through 13-06-2025",
}