@misc{10498/35835, year = {2025}, month = {6}, url = {http://hdl.handle.net/10498/35835}, abstract = {Machine learning-based algorithms have gained wide acceptance over the years due to their high generalization capabilities for a wide range of classification applications. Although these algorithms demonstrate potential and promising performance, they are often limited by speed, particularly when training on large databases. Big data sets with many instances have storage requirements and execution times that can be excessive. This study proposes reducing the sample size generated by the bootstrap resampling method in Ensemble Machine Learning-based models and evaluates their generalization capability on unknown data. Reduced bootstrap samples are employed in the training phase of the Bagging ensemble model. This approach reduces execution times and, consequently, storage requirements. The proposed method was tested on classification tasks, effectively reducing training subset size without compromising performance. Experimental results demonstrate that this approach achieves execution times reductions of up to 70% for some data sets. This reduction has no impact on accuracy, whereas maintaining levels comparable to classical Bagging and its variants. On average, the training subset size was reduced by 25% compared to the original size.}, publisher = {Elsevier}, keywords = {Artificial intelligence}, keywords = {Machine Learning}, keywords = {Big data}, keywords = {Bagging ensemble}, keywords = {Bootstrap}, keywords = {Reduced sample}, title = {REDIBAGG: Reducing the training set size in ensemble machine learning-based prediction models}, doi = {10.1016/j.engappai.2025.110382}, author = {Silva Ramírez, Esther Lydia and Cabrera Sánchez, Juan Francisco and López Coello, Manuel}, }