
@Article{cmc.2026.085321,
AUTHOR = {Adel Assiri},
TITLE = {Large Language Models in Biomedical Text Summarization: A Systematic Review of Architectures, Evaluation Adequacy, and Clinical Readiness},
JOURNAL = {Computers, Materials \& Continua},
VOLUME = {},
YEAR = {},
NUMBER = {},
PAGES = {{pages}},
URL = {http://www.techscience.com/cmc/online/detail/28050},
ISSN = {1546-2226},
ABSTRACT = {The emergence of Large Language Models (LLMs) has transformed biomedical text summarization, shifting research beyond conventional extractive approaches toward increasingly generative and agentic reasoning paradigms. However, rapid advances in model capabilities have exceeded current understanding of their methodological rigor, evaluation adequacy, and clinical readiness. This systematic review conducts a structured exploratory audit of Transformer- and LLM-based biomedical summarization systems to characterize reporting quality, validation practices, and translational maturity. Following PRISMA 2020 guidelines and the Population–Concept–Context framework, we systematically reviewed 178 original English-language studies published between January 2017 and March 2026. A multidimensional evaluation pipeline was developed, incorporating the Methodological Quality Score (MQS) to assess reporting transparency, the Evaluation Adequacy Score (EAS) to quantify validation depth, and the Clinical Readiness Level (CRL) to classify deployment maturity. The field demonstrated substantial recent growth, with 75.8% of included studies published since 2024. Decoder-only architectures represented the most frequently reported model family (50.6%), followed by hybrid or agentic approaches (26.4%). Although reported ROUGE and BERTScore values indicated strong technical performance, substantial heterogeneity across datasets, clinical domains, and summarization tasks limited direct comparison between architecture families. The reviewed studies showed high reporting transparency (median MQS = 7) but limited evaluation depth (median EAS = 0.46). Furthermore, 75.3% of studies remained at laboratory or technical validation stages (CRL 1–2), while only 24.7% achieved institutional feasibility (CRL ≥ 3). Lexical similarity metrics demonstrated limited alignment with clinical readiness, highlighting the need for factuality-centered evaluation and safety-oriented validation strategies. The integrated MQS–EAS–CRL framework provides a structured approach for assessing methodological maturity and translational readiness in biomedical summarization research. Future studies should prioritize robust factuality evaluation, verification mechanisms, and broader clinical validation across diverse healthcare settings.},
DOI = {10.32604/cmc.2026.085321}
}



