
@Article{cmc.2026.084902,
AUTHOR = {Ahmad Raza, Abdul Basit, Syed Muqtar Ahmed, Zeeshan Ahmad Arfeen, Muhammad I. Masud, Muhammad Farid Zamir, Mehreen Kausar Azam, Touqeer Ahmed Jumani},
TITLE = {Vision Transformer–Based Deepfake Detection Across Multiple Generation Methods: A Transfer Learning Approach},
JOURNAL = {Computers, Materials \& Continua},
VOLUME = {},
YEAR = {},
NUMBER = {},
PAGES = {{pages}},
URL = {http://www.techscience.com/cmc/online/detail/27891},
ISSN = {1546-2226},
ABSTRACT = {The development of deepfake technologies is a threat to digital media authentication and cybersecurity infrastructure. The current paper proposes a method for detecting manipulated images of faces based on the Vision Transformer architecture. We fine-tune a pre-trained ViT-Base-Patch16-224 model based on this well-curated dataset of 12,137 face images, which includes an almost equal number of real and synthetic face images using a variety of different generation methods. The data set contains real-life photographs of CelebA and FFHQ, along with artificial samples of the publicly available Kaggle repositories (FaceForensics++, Celeb-DF, and DFDC) and 600 self-collected photos (300 real-life photographs of personal cell phones and 300 artificial ones created with the help of modern tools) to make it closer to real-life use. Methodology involves assessment of the quality of data, systematic preprocessing by ImageNet normalization, data augmentation, and stratification of a 70-20-10 partitioning of the data. AdamW optimization using a learning rate of 2 <mml:math id="mml-ieqn-1"><mml:mo>×</mml:mo></mml:math> 10<sup>–5</sup> was used with 8 epochs. The test accuracy of the system was 99.01%, the precision was 98.85%, and the recall was 99.18%, with 12 misclassifications. The results demonstrate that Vision Transformers can effectively model global image dependencies for detecting Deepfakes. The complete methodology is documented in this paper to ensure reproducibility.},
DOI = {10.32604/cmc.2026.084902}
}



