
@Article{cmc.2026.086229,
AUTHOR = {Milad Moradi},
TITLE = {Safety, Alignment, and Robustness of Large Language Models: A Review},
JOURNAL = {Computers, Materials \& Continua},
VOLUME = {},
YEAR = {},
NUMBER = {},
PAGES = {{pages}},
URL = {http://www.techscience.com/cmc/online/detail/27734},
ISSN = {1546-2226},
ABSTRACT = {Large Language Models (LLMs) have rapidly evolved into general-purpose systems with broad applicability across information access, reasoning, decision support, and human-computer interaction. Their growing deployment, however, has intensified concerns regarding safety, alignment, and robustness, especially as these models become integrated with external tools, retrieval systems, and increasingly agentic workflows. This review provides an analytical overview of the principal risks, technical advances, evaluation practices, and future directions in this area. It first clarifies the conceptual foundations of safety, alignment, robustness, and reliability in the context of LLMs. It then examines the major risk categories associated with LLM deployment, including harmful content generation, hallucination, bias and fairness concerns, privacy leakage, security and misuse risks, and emerging challenges in reasoning-capable and agentic systems. The review further synthesizes recent advances in data-centric safety interventions, post-training alignment, inference-time control, adversarial defense, factuality-oriented safeguards, and system-level protections. It also analyzes the current evaluation landscape, highlighting benchmark fragmentation, limitations of existing methodologies, and the need for more realistic and reproducible assessment. The paper concludes by outlining future research directions toward scalable alignment, stronger real-world evaluation, safer agentic systems, and tighter integration between technical safeguards and governance-oriented approaches.},
DOI = {10.32604/cmc.2026.086229}
}



