{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91717,"databundleVersionId":12184666,"isSourceIdPinned":false,"sourceType":"competition"}],"dockerImageVersionId":30369,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## 📋Table of Contents\n* [Import and First Glance](#import)\n* [Distributions of Features](#eda)\n* [Correlation of Features](#corr)\n* [Target vs Features](#target_features)\n* [Model](#model)","metadata":{}},{"cell_type":"code","source":"# standard\nimport numpy as np\nimport pandas as pd\nimport time\n\n# plots\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly\nimport plotly.express as px\n\n# statistics\nfrom scipy import stats\nfrom sklearn.metrics import cohen_kappa_score\n\n# H2O\nimport h2o\nfrom h2o.estimators import H2OGradientBoostingEstimator\n\n# other stuff\nimport operator","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:53:40.753390Z","iopub.execute_input":"2025-06-02T16:53:40.753835Z","iopub.status.idle":"2025-06-02T16:53:43.924523Z","shell.execute_reply.started":"2025-06-02T16:53:40.753721Z","shell.execute_reply":"2025-06-02T16:53:43.923377Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# configs\npd.set_option('display.max_columns', None) # we want to display all columns in this notebook\npd.set_option('display.max_rows', 100) # increase rows to be displayed\npd.set_option('display.max_colwidth', None) # show full cell contents\n\n# random seed\nmy_random_seed = 111\n\n# aesthetics\ndefault_color_1 = 'darkblue'\ndefault_color_2 = 'darkgreen'\ndefault_color_3 = 'darkred'","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:53:43.925801Z","iopub.execute_input":"2025-06-02T16:53:43.926159Z","iopub.status.idle":"2025-06-02T16:53:43.934137Z","shell.execute_reply.started":"2025-06-02T16:53:43.926127Z","shell.execute_reply":"2025-06-02T16:53:43.932909Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='import'></a>\n# Import and First Glance","metadata":{}},{"cell_type":"code","source":"# load data\ndf_train = pd.read_csv('/kaggle/input/playground-series-s5e6/train.csv')\ndf_test = pd.read_csv('/kaggle/input/playground-series-s5e6/test.csv')\ndf_sub = pd.read_csv('/kaggle/input/playground-series-s5e6/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:53:55.195070Z","iopub.execute_input":"2025-06-02T16:53:55.195446Z","iopub.status.idle":"2025-06-02T16:53:56.520355Z","shell.execute_reply.started":"2025-06-02T16:53:55.195418Z","shell.execute_reply":"2025-06-02T16:53:56.519090Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# preview training data\ndf_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:53:57.239803Z","iopub.execute_input":"2025-06-02T16:53:57.240642Z","iopub.status.idle":"2025-06-02T16:53:57.265432Z","shell.execute_reply.started":"2025-06-02T16:53:57.240603Z","shell.execute_reply":"2025-06-02T16:53:57.264154Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# structure of data - train\ndf_train.info()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:53:59.491646Z","iopub.execute_input":"2025-06-02T16:53:59.493072Z","iopub.status.idle":"2025-06-02T16:53:59.639104Z","shell.execute_reply.started":"2025-06-02T16:53:59.493028Z","shell.execute_reply":"2025-06-02T16:53:59.638026Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 💡 There are no missing values.","metadata":{}},{"cell_type":"code","source":"# structure of data - test\ndf_test.info()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:54:01.553686Z","iopub.execute_input":"2025-06-02T16:54:01.554259Z","iopub.status.idle":"2025-06-02T16:54:01.598659Z","shell.execute_reply.started":"2025-06-02T16:54:01.554213Z","shell.execute_reply":"2025-06-02T16:54:01.597378Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='eda'></a>\n# Distributions of Features","metadata":{}},{"cell_type":"code","source":"# basic stats - train\ndf_train.describe(include='all')","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:54:03.760435Z","iopub.execute_input":"2025-06-02T16:54:03.760798Z","iopub.status.idle":"2025-06-02T16:54:04.254471Z","shell.execute_reply.started":"2025-06-02T16:54:03.760769Z","shell.execute_reply":"2025-06-02T16:54:04.253318Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# basic stats - test\ndf_test.describe(include='all')","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:54:06.345879Z","iopub.execute_input":"2025-06-02T16:54:06.346297Z","iopub.status.idle":"2025-06-02T16:54:06.505246Z","shell.execute_reply.started":"2025-06-02T16:54:06.346263Z","shell.execute_reply":"2025-06-02T16:54:06.504117Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# define features and target\n\n# numerical\nfeatures_num = ['Temparature', 'Humidity', 'Moisture', \n                'Nitrogen', 'Potassium', 'Phosphorous']\n\n# categorical\nfeatures_cat = ['Soil Type', 'Crop Type']\n\n# target\ntarget = 'Fertilizer Name'","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:54:08.397955Z","iopub.execute_input":"2025-06-02T16:54:08.398368Z","iopub.status.idle":"2025-06-02T16:54:08.404746Z","shell.execute_reply.started":"2025-06-02T16:54:08.398336Z","shell.execute_reply":"2025-06-02T16:54:08.403270Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot histograms (train and test)\nfor f in features_num:\n    plt.figure(figsize=(14,3))\n    ax1 = plt.subplot(1,2,1)\n    df_train[f].plot(kind='hist', bins=25, color=default_color_1)\n    plt.title(f + ' - Train')\n    plt.grid()\n    ax2 = plt.subplot(1,2,2, sharex=ax1)\n    df_test[f].plot(kind='hist', bins=25, color=default_color_2)\n    plt.title(f + ' - Test')\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:54:10.282467Z","iopub.execute_input":"2025-06-02T16:54:10.283037Z","iopub.status.idle":"2025-06-02T16:54:13.367439Z","shell.execute_reply.started":"2025-06-02T16:54:10.282970Z","shell.execute_reply":"2025-06-02T16:54:13.366360Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# boxplots (train and test)\nfor f in features_num:\n    plt.figure(figsize=(14,1))\n    ax1 = plt.subplot(1,2,1)\n    df_temp = df_train[f].dropna() # boxplot does not like missings...\n    plt.boxplot(df_temp, vert=False)\n    plt.title(f + ' - Train')\n    plt.grid()\n    ax2 = plt.subplot(1,2,2, sharex=ax1)\n    df_temp = df_test[f].dropna()\n    plt.boxplot(df_temp, vert=False)\n    plt.title(f + ' - Test')\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:54:15.654685Z","iopub.execute_input":"2025-06-02T16:54:15.655147Z","iopub.status.idle":"2025-06-02T16:54:17.151601Z","shell.execute_reply.started":"2025-06-02T16:54:15.655112Z","shell.execute_reply":"2025-06-02T16:54:17.150473Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot categorical feature distributions (train and test)\nfor f in features_cat:\n    plt.figure(figsize=(15,3))\n    ax1 = plt.subplot(1,2,1)\n    df_train[f].value_counts().sort_index().plot(kind='bar', color=default_color_1)\n    plt.title(f + ' - Train')\n    plt.grid()\n    ax2 = plt.subplot(1,2,2)\n    df_test[f].value_counts().sort_index().plot(kind='bar', color=default_color_2)\n    plt.title(f + ' - Test')\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:54:18.898612Z","iopub.execute_input":"2025-06-02T16:54:18.899058Z","iopub.status.idle":"2025-06-02T16:54:19.741956Z","shell.execute_reply.started":"2025-06-02T16:54:18.898984Z","shell.execute_reply":"2025-06-02T16:54:19.740801Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='corr'></a>\n# Correlation of Features","metadata":{}},{"cell_type":"code","source":"# calc correlation matrices\ncorr_pearson = df_train[features_num].corr(method='pearson')\ncorr_spearman = df_train[features_num].corr(method='spearman')","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:54:21.687546Z","iopub.execute_input":"2025-06-02T16:54:21.687949Z","iopub.status.idle":"2025-06-02T16:54:22.631119Z","shell.execute_reply.started":"2025-06-02T16:54:21.687915Z","shell.execute_reply":"2025-06-02T16:54:22.629928Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# and plot them\nplt.figure(figsize=(6,5))\nsns.heatmap(corr_pearson, annot=True, cmap='RdYlGn',\n            fmt='.3f', linecolor='black', linewidths=0.5,\n            vmin=-1, vmax=+1)\nplt.title('Pearson Correlation')\n\nplt.figure(figsize=(6,5))\nsns.heatmap(corr_spearman, annot=True, cmap='RdYlGn', \n            fmt='.3f', linecolor='black', linewidths=0.5,\n            vmin=-1, vmax=+1)\nplt.title('Spearman Correlation')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:54:23.695779Z","iopub.execute_input":"2025-06-02T16:54:23.696195Z","iopub.status.idle":"2025-06-02T16:54:24.549061Z","shell.execute_reply.started":"2025-06-02T16:54:23.696160Z","shell.execute_reply":"2025-06-02T16:54:24.547877Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### 💡 All features basically uncorrelated. Smells like artificially created data.","metadata":{}},{"cell_type":"markdown","source":"<a id='target_features'></a>\n# Target vs Features","metadata":{}},{"cell_type":"code","source":"# plot target distribution\ndf_train[target].value_counts().sort_index().plot(kind='bar', color=default_color_3)\nplt.title('Target')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:54:31.217194Z","iopub.execute_input":"2025-06-02T16:54:31.217552Z","iopub.status.idle":"2025-06-02T16:54:31.468296Z","shell.execute_reply.started":"2025-06-02T16:54:31.217518Z","shell.execute_reply":"2025-06-02T16:54:31.467085Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#### Not to unbalanced, we do not need to use any up-/down-sampling tricks here.","metadata":{}},{"cell_type":"code","source":"# target vs numerical features\nfor f in features_num:\n    plt.figure(figsize=(8,4))\n    sns.violinplot(data=df_train, x=target, y=f)\n    plt.title('Target vs ' + f)\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:54:33.328824Z","iopub.execute_input":"2025-06-02T16:54:33.329257Z","iopub.status.idle":"2025-06-02T16:54:46.468226Z","shell.execute_reply.started":"2025-06-02T16:54:33.329221Z","shell.execute_reply":"2025-06-02T16:54:46.466877Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# impact of categorical features - normalized cross tables\nfor f in features_cat:\n    ctab = pd.crosstab(df_train[target], df_train[f])\n    ctab_norm = ctab / ctab.sum()\n    plt.figure(figsize=(12,3))\n    g = sns.heatmap(ctab_norm, annot=True,\n                    fmt='.2%', linecolor='black',\n                    linewidths=1, cmap='Greens', \n                    vmin=0.1, vmax=0.2)\n    plt.title(f + ' vs target - train')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T17:10:02.481100Z","iopub.execute_input":"2025-06-02T17:10:02.482139Z","iopub.status.idle":"2025-06-02T17:10:03.788087Z","shell.execute_reply.started":"2025-06-02T17:10:02.482101Z","shell.execute_reply":"2025-06-02T17:10:03.786785Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<a id='model'></a>\n# Model","metadata":{}},{"cell_type":"code","source":"# start H2O\nh2o.init(max_mem_size='12G', nthreads=4)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2025-06-02T16:54:51.993778Z","iopub.execute_input":"2025-06-02T16:54:51.995145Z","iopub.status.idle":"2025-06-02T16:55:00.183450Z","shell.execute_reply.started":"2025-06-02T16:54:51.995090Z","shell.execute_reply":"2025-06-02T16:55:00.182095Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# prepare data for upload in H2O environment\ny = df_train[target]\n\ndf_train['fold'] = 'train'\ndf_test['fold'] = 'test'\n\ndf_train[target] = 'x_' + y # add prefix to avoid problems with predictor names\ndf_test[target] = 0\n\n# combine train & test in one data frame\ndf = pd.concat([df_train, df_test])","metadata":{"execution":{"iopub.status.busy":"2025-06-02T16:55:00.185532Z","iopub.execute_input":"2025-06-02T16:55:00.185893Z","iopub.status.idle":"2025-06-02T16:55:00.511250Z","shell.execute_reply.started":"2025-06-02T16:55:00.185848Z","shell.execute_reply":"2025-06-02T16:55:00.509884Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# upload\ndf_hex = h2o.H2OFrame(df)\n# and split again\ntrain_hex = df_hex[df_hex['fold']=='train']\ntest_hex = df_hex[df_hex['fold']=='test']","metadata":{"execution":{"iopub.status.busy":"2025-06-02T17:10:41.561096Z","iopub.execute_input":"2025-06-02T17:10:41.561491Z","iopub.status.idle":"2025-06-02T17:10:49.063855Z","shell.execute_reply.started":"2025-06-02T17:10:41.561458Z","shell.execute_reply":"2025-06-02T17:10:49.062699Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# force categorical target\ntrain_hex[target] = train_hex[target].asfactor()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T17:10:52.424545Z","iopub.execute_input":"2025-06-02T17:10:52.424935Z","iopub.status.idle":"2025-06-02T17:10:52.705901Z","shell.execute_reply.started":"2025-06-02T17:10:52.424899Z","shell.execute_reply":"2025-06-02T17:10:52.704849Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# define predictors\npredictors = features_num + features_cat\nprint(predictors)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T17:10:53.082415Z","iopub.execute_input":"2025-06-02T17:10:53.082801Z","iopub.status.idle":"2025-06-02T17:10:53.088718Z","shell.execute_reply.started":"2025-06-02T17:10:53.082768Z","shell.execute_reply":"2025-06-02T17:10:53.087413Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# define model\nn_CV = 5\ngbm_model = H2OGradientBoostingEstimator(distribution = 'multinomial',\n                                         nfolds = n_CV,\n                                         ntrees = 400,\n                                         learn_rate = 0.01,\n                                         max_depth = 9,\n                                         col_sample_rate = 0.7,                                    \n                                         stopping_rounds = 10,\n                                         stopping_tolerance = 0.0001,\n                                         stopping_metric = 'log_loss',\n                                         score_each_iteration = True,                                          \n                                         seed=my_random_seed)\n\n# and train model\nt1 = time.time()\ngbm_model.train(predictors, target, training_frame = train_hex);\nt2 = time.time()\nprint('Elapsed time [s]:', np.round(t2-t1,4))","metadata":{"execution":{"iopub.status.busy":"2025-06-02T17:18:17.657773Z","iopub.execute_input":"2025-06-02T17:18:17.658227Z","iopub.status.idle":"2025-06-02T17:20:10.500665Z","shell.execute_reply.started":"2025-06-02T17:18:17.658190Z","shell.execute_reply":"2025-06-02T17:20:10.499447Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# short summary of model\ngbm_model.show_summary()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T17:20:12.750360Z","iopub.execute_input":"2025-06-02T17:20:12.750777Z","iopub.status.idle":"2025-06-02T17:20:12.759470Z","shell.execute_reply.started":"2025-06-02T17:20:12.750743Z","shell.execute_reply":"2025-06-02T17:20:12.758413Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# show scoring history - training vs cross validations\nfor i in range(n_CV):\n    cv_model_temp = gbm_model.cross_validation_models()[i]\n    df_cv_score_history = cv_model_temp.score_history()\n    my_title = 'CV ' + str(1+i) + ' - Scoring History [log_loss]'\n    plt.scatter(df_cv_score_history.number_of_trees,\n                y=df_cv_score_history.training_logloss, \n                c='blue', label='training')\n    plt.scatter(df_cv_score_history.number_of_trees,\n                y=df_cv_score_history.validation_logloss, \n                c='darkorange', label='validation')\n    plt.title(my_title)\n    plt.xlabel('Number of Trees')\n    plt.legend()\n    plt.grid()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-02T17:20:15.164727Z","iopub.execute_input":"2025-06-02T17:20:15.165177Z","iopub.status.idle":"2025-06-02T17:20:15.905096Z","shell.execute_reply.started":"2025-06-02T17:20:15.165140Z","shell.execute_reply":"2025-06-02T17:20:15.903696Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# show cross validation results\ngbm_model.cross_validation_metrics_summary().as_data_frame()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T17:20:24.061053Z","iopub.execute_input":"2025-06-02T17:20:24.061441Z","iopub.status.idle":"2025-06-02T17:20:24.077538Z","shell.execute_reply.started":"2025-06-02T17:20:24.061411Z","shell.execute_reply":"2025-06-02T17:20:24.076263Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# variable importance\ngbm_model.varimp_plot(10);","metadata":{"execution":{"iopub.status.busy":"2025-06-02T17:16:42.243103Z","iopub.execute_input":"2025-06-02T17:16:42.243497Z","iopub.status.idle":"2025-06-02T17:16:42.454226Z","shell.execute_reply.started":"2025-06-02T17:16:42.243465Z","shell.execute_reply":"2025-06-02T17:16:42.453043Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Evaluate on training set","metadata":{}},{"cell_type":"code","source":"# predict on training data\npred_train = gbm_model.predict(train_hex)\npred_train = pred_train.as_data_frame();\npred_train.head()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T17:08:52.451616Z","iopub.execute_input":"2025-06-02T17:08:52.452537Z","iopub.status.idle":"2025-06-02T17:09:01.120705Z","shell.execute_reply.started":"2025-06-02T17:08:52.452500Z","shell.execute_reply":"2025-06-02T17:09:01.119560Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# summary of predictions\nprint(pred_train.predict.value_counts())\npred_train.predict.value_counts().plot(kind='bar', color=default_color_3)\nplt.title('Predictions - Train')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T17:09:04.194171Z","iopub.execute_input":"2025-06-02T17:09:04.194751Z","iopub.status.idle":"2025-06-02T17:09:04.438766Z","shell.execute_reply.started":"2025-06-02T17:09:04.194700Z","shell.execute_reply":"2025-06-02T17:09:04.437609Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# confusion matrix\nconf_train = pd.crosstab(pred_train.predict, df_train[target])\nsns.heatmap(conf_train, annot=True, cmap='Reds', \n            fmt='.0f', linecolor='black', linewidths=0.5)\nplt.title('Confusion Matrix - Training')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T17:09:07.605782Z","iopub.execute_input":"2025-06-02T17:09:07.606202Z","iopub.status.idle":"2025-06-02T17:09:08.148943Z","shell.execute_reply.started":"2025-06-02T17:09:07.606168Z","shell.execute_reply":"2025-06-02T17:09:08.147872Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Evaluate on test set","metadata":{}},{"cell_type":"code","source":"# predict on test data\npred_test = gbm_model.predict(test_hex)\npred_test = pred_test.as_data_frame();\npred_test.head()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T17:09:17.022896Z","iopub.execute_input":"2025-06-02T17:09:17.023345Z","iopub.status.idle":"2025-06-02T17:09:21.220974Z","shell.execute_reply.started":"2025-06-02T17:09:17.023311Z","shell.execute_reply":"2025-06-02T17:09:21.219833Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# summary of predictions\nprint(pred_test.predict.value_counts())\npred_test.predict.value_counts().plot(kind='bar', color=default_color_3)\nplt.title('Predictions - Test')\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-06-02T17:09:22.474962Z","iopub.execute_input":"2025-06-02T17:09:22.476112Z","iopub.status.idle":"2025-06-02T17:09:22.699160Z","shell.execute_reply.started":"2025-06-02T17:09:22.476069Z","shell.execute_reply":"2025-06-02T17:09:22.697918Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Extract top 3 labels for each observation","metadata":{}},{"cell_type":"markdown","source":"#### Credits to https://www.kaggle.com/code/satyaprakashshukl/predicting-optimal-fertilizers for inspiration how to tackle this","metadata":{}},{"cell_type":"code","source":"# first get labels in order given by prediction data frame\nlabels = pred_test.columns[2:9].tolist()\nlabels = [x[2:] for x in labels] # remove prefix again\nprint(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T15:08:29.003545Z","iopub.execute_input":"2025-06-01T15:08:29.004004Z","iopub.status.idle":"2025-06-01T15:08:29.011775Z","shell.execute_reply.started":"2025-06-01T15:08:29.003958Z","shell.execute_reply":"2025-06-01T15:08:29.010310Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# get top 3 indices for each row using \"argsort\"\ntest_probs = np.asarray(pred_test.iloc[:,2:9])\ntop_3_preds = np.argsort(test_probs, axis=1)[:, -3:][:, ::-1]\ntop_3_preds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T15:08:29.013449Z","iopub.execute_input":"2025-06-01T15:08:29.013914Z","iopub.status.idle":"2025-06-01T15:08:29.061310Z","shell.execute_reply.started":"2025-06-01T15:08:29.013849Z","shell.execute_reply":"2025-06-01T15:08:29.060218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# flatten array to a simple array of indices\nflat_indices = top_3_preds.ravel()\nflat_indices","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T15:08:29.062996Z","iopub.execute_input":"2025-06-01T15:08:29.063506Z","iopub.status.idle":"2025-06-01T15:08:29.074265Z","shell.execute_reply.started":"2025-06-01T15:08:29.063453Z","shell.execute_reply":"2025-06-01T15:08:29.073143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extract label items according to this integer array\ngetter = operator.itemgetter(*flat_indices)\nlabel_list = list(getter(labels))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T15:08:29.075535Z","iopub.execute_input":"2025-06-01T15:08:29.075863Z","iopub.status.idle":"2025-06-01T15:08:29.153249Z","shell.execute_reply.started":"2025-06-01T15:08:29.075833Z","shell.execute_reply":"2025-06-01T15:08:29.151964Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# group in triples again\ntop_3_labels = np.asarray(label_list).reshape(top_3_preds.shape)\ntop_3_labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T15:08:29.154514Z","iopub.execute_input":"2025-06-01T15:08:29.155015Z","iopub.status.idle":"2025-06-01T15:08:29.300985Z","shell.execute_reply.started":"2025-06-01T15:08:29.154864Z","shell.execute_reply":"2025-06-01T15:08:29.299828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# final step: create prediction strings\npred_string = [' '.join(row) for row in top_3_labels]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-01T15:08:29.302284Z","iopub.execute_input":"2025-06-01T15:08:29.302605Z","iopub.status.idle":"2025-06-01T15:08:29.828330Z","shell.execute_reply.started":"2025-06-01T15:08:29.302576Z","shell.execute_reply":"2025-06-01T15:08:29.826604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# submission\ndf_sub[target] = pred_string\ndf_sub.head(10)","metadata":{"execution":{"iopub.status.busy":"2025-06-01T15:08:29.830024Z","iopub.execute_input":"2025-06-01T15:08:29.830486Z","iopub.status.idle":"2025-06-01T15:08:29.861910Z","shell.execute_reply.started":"2025-06-01T15:08:29.830440Z","shell.execute_reply":"2025-06-01T15:08:29.860790Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# save file\ndf_sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2025-06-01T15:08:29.863343Z","iopub.execute_input":"2025-06-01T15:08:29.863746Z","iopub.status.idle":"2025-06-01T15:08:30.264058Z","shell.execute_reply.started":"2025-06-01T15:08:29.863703Z","shell.execute_reply":"2025-06-01T15:08:30.262944Z"},"trusted":true},"outputs":[],"execution_count":null}]}