
@Article{cmc.2026.085750,
AUTHOR = {Hung-Jr Shiu, Ming-Ya Tseng, Wei-Chung Lin},
TITLE = {EchoMark: A Practical Audio Disruption Scheme for Anti-Synthesis Protection},
JOURNAL = {Computers, Materials \& Continua},
VOLUME = {},
YEAR = {},
NUMBER = {},
PAGES = {{pages}},
URL = {http://www.techscience.com/cmc/online/detail/28052},
ISSN = {1546-2226},
ABSTRACT = {Driven by recent breakthroughs in generative artificial intelligence, modern voice cloning technologies can synthesize remarkably lifelike human speech, exacerbating security vulnerabilities associated with identity impersonation, financial fraud, and deepfake audio proliferation. To mitigate these risks, this paper introduces EchoMark, an acoustic-layer disruption framework designed to systematically undermine neural speech generation workflows. Unlike conventional digital watermarking or software-level perturbation strategies, EchoMark embeds structured, multi-tiered echo patterns directly into audio during physical playback and re-recording. This physical-layer integration severely compromises the spectral coherence essential for neural text-to-speech (TTS) modeling, resulting in degraded acoustic fidelity and impaired speech-fitting capabilities. We rigorously validate EchoMark across multiple representative TTS architectures—specifically Mockingbird, FishSpeech, and GPT-SoVITS—under diverse operating environments, dataset categories, and quantitative metrics. Experimental evaluations confirm that structured echo perturbations serve as an efficient, lightweight physical countermeasure capable of inducing performance degradation in target speech synthesis pipelines, providing a viable anti-spoofing defense for physical-layer audio applications.},
DOI = {10.32604/cmc.2026.085750}
}



