
@Article{cmes.2026.085393,
AUTHOR = {Mrugendrasinh Rahevar, Martin Parmar, Hemant Yadav, Chun-Ta Li, Agbotiname Lucky Imoize, Hiren Mewada},
TITLE = {Quantum Kernels for Text Classification: A Statistical and Diagnostic Framework Revealing the Low-Data Regime},
JOURNAL = {Computer Modeling in Engineering \& Sciences},
VOLUME = {148},
YEAR = {2026},
NUMBER = {2},
PAGES = {--},
URL = {http://www.techscience.com/CMES/v148n2/68580},
ISSN = {1526-1506},
ABSTRACT = {Quantum kernel techniques aim to leverage quantum computational capabilities on social data. However, their application to natural language processing tasks faces formidable obstacles, such as extreme dimensionality reduction (<math id="mml-ieqn-1"><mi>D</mi><mo>=</mo><mn>384</mn><mo stretchy="false">→</mo><mi>k</mi><mo>=</mo><mn>8</mn></math>), concentration of measure in quantum feature spaces, and the lack of theoretical understanding of when quantum advantages occur in kernel-based text classification. Filling this gap, we provide a comprehensive study of quantum kernels for text classification that addresses three major challenges in existing studies: general data compression approaches that ignore class structure, the lack of a predictive diagnostic toolkit, and overlooked approaches for handling concentration effects. Our main contributions include a supervised contrastive data compression approach with theoretical guarantees of kernel alignment, a five-diagnostic toolkit connecting theoretical insights with practical performance, the discovery of a small-data regime (<math id="mml-ieqn-2"><mi>n</mi><mo>≤</mo><mn>200</mn></math>) in which quantum methods perform comparably to classical approaches, and a transparent demonstration that quantum kernels require carefully designed settings to remain competitive with classical counterparts. Across four datasets, evaluated using five random seeds and exact paired statistical testing, we observe that quantum projected kernels achieve an accuracy of <math id="mml-ieqn-3"><mn>84.5</mn><mi mathvariant="normal">%</mi></math> compared to <math id="mml-ieqn-4"><mn>85.1</mn><mi mathvariant="normal">%</mi></math> for the classical RBF kernel (<math id="mml-ieqn-5"><mi>p</mi><mo>&gt;</mo><mn>0.05</mn></math>, not statistically significant) under supervised compression settings. However, quantum methods lag behind by approximately <math id="mml-ieqn-6"><mn>2</mn></math>%–<math id="mml-ieqn-7"><mn>4</mn><mi mathvariant="normal">%</mi></math> at larger scales due to concentration effects, reflected in reduced off-diagonal kernel variance (<math id="mml-ieqn-8"><mn>0.003</mn></math> vs. <math id="mml-ieqn-9"><mn>0.025</mn></math>).},
DOI = {10.32604/cmes.2026.085393}
}



