
@Article{jai.2026.085269,
AUTHOR = {Richard Adusei, Gaddafi Abdul-Salaam},
TITLE = {A Review of Fine-Grained Visual Categorization with Deep Learning},
JOURNAL = {Journal on Artificial Intelligence},
VOLUME = {8},
YEAR = {2026},
NUMBER = {1},
PAGES = {425--472},
URL = {http://www.techscience.com/jai/v8n1/68750},
ISSN = {2579-003X},
ABSTRACT = {Fine-grained visual categorization (FGVC) presents a class of recognition problems in which the discriminative signal is spatially concentrated, visually subtle, and easily destroyed by the preprocessing and augmentation strategies that serve coarse recognition well. Where standard image classification requires a model to distinguish birds from cars, FGVC requires it to distinguish one bird species from another, a task that demands localization, feature-space shaping, and representation learning to operate in close coordination. This survey synthesizes thirty recent works spanning 2021 to 2026, organizing them under five interlocking themes, namely discriminative region discovery, metric learning and loss function design, data augmentation and training strategy, architectural and multi-scale feature representation, and foundation model adaptation. Rather than presenting these themes as independent research tracks, the survey examines them as overlapping responses to a single underlying difficulty, the impossibility of reliably separating where to look from how to represent what is found. Key tensions are identified across the literature between single-image and cross-image localization strategies, between background suppression and foreground discovery, between image-level and feature-level augmentation including generative diffusion augmentation, between geometrically principled and engineering-scale contrastive losses, and between the scale offered by foundation models and the architectural conservatism of purpose-built FGVC designs. The synthesis reveals that the most capable recent systems including FG-CLIP, FEM, HI2R, SIM-OFE, and CSDNet resolve these tensions through explicit integration of localization and feature-space objectives, that ultra-FGVC at the cultivar level is emerging as a distinct and harder sub-problem requiring dedicated solutions, and that the field’s principal open problems including occlusion robustness, cross-dataset generalization, and the theoretical relationship between fine-grained discriminability and low-rank representation remain largely unaddressed.},
DOI = {10.32604/jai.2026.085269}
}



