@article {10.3844/jcssp.2026.2711.2722, article_type = {journal}, title = {An Active Learning Ensemble Framework for Tropical Crop Yield Prediction}, author = {Bhilare, Amol and Swain, Debabrata and Patel, Megh and Mishra, Sashikala and Kumar, Nitesh and Kumar, Manish}, volume = {22}, number = {9}, year = {2026}, month = {Sep}, pages = {2711-2722}, doi = {10.3844/jcssp.2026.2711.2722}, url = {https://thescipub.com/abstract/jcssp.2026.2711.2722}, abstract = {Agriculture always has great importance among all different sectors, not only in the world but also in India. It has a significant impact on food security and largely controls the economy. At present, technological advancements have significantly enhanced agricultural productivity, but it can still be further improved by the integration of new cutting-edge techniques like Artificial intelligence. Among the factors that help farmers decide which crop to cultivate, the expected yield of the crop in a given environment is one of the most important. There is therefore a clear need for an AI-based prediction system that can assist farmers in this process. The proposed system uses a hybrid stacked ensemble regression method, combining XGBoost and Random Forest as base learners and Ridge Regression as the meta-learner, to predict crop yield. It aims to enhance agricultural productivity and decision-making by integrating agricultural, meteorological, and soil data with advanced analytics. To train the models efficiently, an active learning strategy is employed that selects informative data points by balancing both diversity and uncertainty, thereby reducing the labelled-data requirement. Hyperparameter tuning is performed using GridSearchCV with cross-validation. The unique contribution of this work lies in coupling a stacked ensemble with a hybrid uncertainty-and-diversity active learning query strategy over a large multi-source tropical-crop dataset that jointly integrates spatial, temporal, climatic, and soil variables. On the test set, the hybrid model attained a coefficient of determination (R²) of 0.96 without active learning and 0.97 with active learning, and was further evaluated using complementary error metrics (MAE and RMSE) and statistical significance testing to support the reliability of the reported gains.}, journal = {Journal of Computer Science}, publisher = {Science Publications} }