
@Article{cmes.2026.082841,
AUTHOR = {Yiyan Zhang, Yi Xin, Qin Li},
TITLE = {Quantitative Profiling of Tabular Biomedical Benchmark Datasets: A Meta-Learning Perspective for Algorithm Selection},
JOURNAL = {Computer Modeling in Engineering \& Sciences},
VOLUME = {},
YEAR = {},
NUMBER = {},
PAGES = {{pages}},
URL = {http://www.techscience.com/CMES/online/detail/27596},
ISSN = {1526-1506},
ABSTRACT = {Medical data has specificity compared to other fields of data, and the description of medical data characteristics is still in a qualitative stage. This study included 293 sub-datasets of 138 independent datasets. First, data preprocessing was performed using methods such as incomplete data removal, inconsistent data normalization, and data integration. Then, the characteristics of 293 research datasets were quantified using 26 indicators in three categories: simple indicators, statistical indicators, and informational indicators. Furthermore, statistical analysis was performed on the above-mentioned quantitative characteristics, and stepwise regression and decision tree methods were used for modeling learning. The characteristics of the biological and medical datasets in the study were compared with those of other fields’ datasets. By comparing the results of statistical analysis and learning modeling, the study found that the sample size of medical datasets included in the UCI database analyzed in this paper is small, most within 1000. The harmonic mean or geometric mean of continuous variables is significantly higher than the data from other fields. That is to say, the scope of the continuous variable range is large. This study uses quantitative indicators to describe the characteristics of medical datasets to avoid the decrease in credibility caused by subjective analysis, and lays a foundation for further algorithm applicability research.},
DOI = {10.32604/cmes.2026.082841}
}



