@article{4825, author = {Martin Lopez Nores}, title = {Predictive Modelling of Overall Performance in IoT-Enabled Multicultural English Learning Environments: An Ensemble Learning and Interpretability Analysis}, journal = {Digital Signal Processing and Artificial Intelligence for Automatic Learning}, year = {2026}, volume = {5}, number = {3}, doi = {https://doi.org/10.6025/dspaial/2026/5/3/187-208}, url = {https://www.dline.info/dspai/fulltext/v5n3/dspaiv5n3_3.pdf}, abstract = {The convergence of Internet of Things (IoT) technologies, artificial intelligence, and educational data analytics has generated considerable interest in personalised English language instruction within multicultural learning environments. However, the extent to which IoT-derived behavioural, linguistic, and engagement features can reliably predict overall learner performance remains insufficiently examined, particularly with respect to model transparency and pedagogical interpretability. This study conducts a comprehensive predictive analysis of Overall_Performance using the English Learning Interaction Dataset, comprising 15,000 learningsession records and 54 features collected from IoT-enabled smart classrooms at Chinese universities. Three ensemble tree-based regressors (Gradient Boosting, Random Forest, and Extra Trees) were trained to predict continuous performance outcomes, while Random Forest and Logistic Regression classifiers were applied to a dichotomised high-performance target (  75). Additionally, one-way analysis of variance (ANOVA) was employed to compare mean performance across six teaching strategies. Model interpretability was assessed using SHapley Additive exPlanations (SHAP), Partial Dependence Plots (PDP), and Individual Conditional Expectation (ICE) curves. Results revealed negligible predictive utility across all models: ensemble regressors yielded R² values at or below zero (best: Gradient Boosting, R² = - 0.0022; MAE  15.0; RMSE  17.3), and binary classifiers produced ROC-AUC values indistinguishable from chance (  0.50). One-way ANOVA detected no statistically significant difference in Overall_Performance among the six teaching strategies (F = 0.237, p = 0.946). Despite weak overall predictivity, SHAP and feature-importance analyses consistently identified Communication_Effectiveness, Pronunciation_Score, Assignment_Marks, Feedback_Score, and Grammar_Score as the relatively strongest predictors, while PDP/ICE plots revealed flat average marginal effects accompanied by substantial between learner heterogeneity. These findings indicate that the current tabular feature set is insufficient to predict overall performance with practical accuracy or to differentiate among pedagogical approaches. The study delivers a “Tabular Gradient Boosting” analysis and highlights the need for longitudinal modelling frameworks and alternative outcome metrics to realise the full potential of AI-driven personalised English education.}, }