{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":11906782,"sourceType":"datasetVersion","datasetId":7485036}],"dockerImageVersionId":30746,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"72b3f825","cell_type":"markdown","source":"# **Random Forest**","metadata":{}},{"id":"331815e4","cell_type":"markdown","source":"### **Why Random Forest?**\nAs initial model, I've choosen the Random Forest classier due to its robustness and performance advantages:\n- it exploit bagging to aggregate multiple fully-grown decision trees, which reduce variance and helps prevent overfitting.\n- each tree is built on a small subset of the features, increasing the diversity among trees -> improves generalization.","metadata":{}},{"id":"f604f43d","cell_type":"markdown","source":"### **Data Preparation**\nThis part consist of what we did in the first notebook.","metadata":{}},{"id":"98c19ee6","cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.preprocessing import OneHotEncoder, StandardScaler\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n\ndf_train = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\n\n#### DATA PREPARATION\n\ntrain_features = df_train.columns.tolist()\ntest_features = df_test.columns.tolist()\nfeatures_toremove =  list(set(train_features) - set(test_features) - {'sii'})\n\ndel df_train['id']\n\nfor col in features_toremove:\n    del df_train[col]\n\ndf_train.dropna(subset=['sii'], inplace=True)\n\ndf_train = df_train[df_train['Physical-BMI']!=0]\ndf_train = df_train[df_train['Physical-Weight']!=0]\n\nphysical_measures_df = pd.read_csv('/kaggle/input/nhanes-physical-measures/physical_measures.csv')\n\ndf_train = df_train.merge( physical_measures_df, on=['Basic_Demos-Age', 'Basic_Demos-Sex'], suffixes=('', '_avg'))\ncols = ['Physical-BMI','Physical-Height','Physical-Weight','Physical-Waist_Circumference','Physical-Diastolic_BP','Physical-HeartRate','Physical-Systolic_BP']\ntot_nan_phys = df_train[cols].isna().all(axis=1)\n\nfor col in cols:\n    df_train.loc[tot_nan_phys, col] = df_train.loc[tot_nan_phys, f\"{col}_avg\"]\n    del df_train[f\"{col}_avg\"]\n\nthreshold = int(df_train.shape[1] * 0.35)\ndf_train.dropna(thresh=threshold, inplace=True)\n\nX = df_train.iloc[:, :-1]\ny = df_train.iloc[:, -1]\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.20, random_state=42)\n\nis_numerical = np.array([np.issubdtype(dtype, np.number) for dtype in X.dtypes])  \nnumerical_idx = np.flatnonzero(is_numerical) \nnew_X_train = X_train.iloc[:, numerical_idx]\nnew_X_test = X_test.iloc[:, numerical_idx]\n\n\nscaler = StandardScaler()\nimputer = KNNImputer(n_neighbors=3)\n\nscaled_train = scaler.fit_transform(new_X_train)\nX_array = imputer.fit_transform(scaled_train)\nX_array = scaler.inverse_transform(X_array)\nnew_X_train = pd.DataFrame(X_array, columns=new_X_train.columns, index=new_X_train.index) # convert into a dataframe since X_array is of type ndarray\n\nscaled_test = scaler.fit_transform(new_X_test)\nX_array = imputer.fit_transform(scaled_test)\nX_array = scaler.inverse_transform(X_array)\nnew_X_test = pd.DataFrame(X_array, columns=new_X_test.columns, index=new_X_test.index)\n\ncategorical_idx = np.flatnonzero(is_numerical==False)\ncategorical_X_train = X_train.iloc[:, categorical_idx]\ncategorical_X_test = X_test.iloc[:, categorical_idx]\n\nimputer = SimpleImputer(strategy='most_frequent')\nX_array = imputer.fit_transform(categorical_X_train)\ncategorical_X_train = pd.DataFrame(X_array, columns=categorical_X_train.columns, index=categorical_X_train.index)\n\nX_array = imputer.fit_transform(categorical_X_test)\ncategorical_X_test = pd.DataFrame(X_array, columns=categorical_X_test.columns, index=categorical_X_test.index)\n\n\noh = OneHotEncoder(sparse_output=False)\n\noh.fit(categorical_X_train)\nencoded = oh.transform(categorical_X_train)\n\nfor i, col in enumerate(oh.get_feature_names_out()):\n    new_X_train = new_X_train.copy()\n    new_X_train[col] = encoded[:, i]\n\noh.fit(categorical_X_test)\nencoded = oh.transform(categorical_X_test)\n\nfor i, col in enumerate(oh.get_feature_names_out()):\n    new_X_test = new_X_test.copy()\n    new_X_test[col] = encoded[:, i]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T10:36:34.704725Z","iopub.execute_input":"2025-05-22T10:36:34.705146Z","iopub.status.idle":"2025-05-22T10:36:36.211328Z","shell.execute_reply.started":"2025-05-22T10:36:34.705114Z","shell.execute_reply":"2025-05-22T10:36:36.209926Z"}},"outputs":[],"execution_count":null},{"id":"c0f8fe7b","cell_type":"markdown","source":"### **Tuning the HyperParameter**\nWe have the Random Forest model from the previous notebook as basic model, let's do some refinement and see if it's possible to improve our accuracy. <br>\nTo do this we will tune the hyperparameter of the Random Forest model.<br>\nUsually the validation set is used to do this, indeed the validation set is used to simulate an unseen test set on which it's possible to tune/validate the algorithm's parameters. <br>\nWe are going to use a k-fold cross-validation that tune the hyperparameters automatically.","metadata":{}},{"id":"a4da5b30","cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import accuracy_score\n\nbase_model = RandomForestClassifier()\nparameters = { 'n_estimators': [50, 100, 200],\n    'max_leaf_nodes': [50, 80, 100],\n    'max_depth': [10, 20, None],\n    'min_samples_split': [2, 5, 10],\n    'bootstrap': [True, False]\n    }\ntuned_model = GridSearchCV(base_model, parameters, cv=5, scoring='accuracy', n_jobs=-1)\ntuned_model.fit(new_X_train, y_train)\nprint (\"Best Score: {:.3f}\".format(tuned_model.best_score_) )\nprint(\"Best Params: \", tuned_model.best_params_)\ntest_acc = accuracy_score(y_true = y_test, y_pred = tuned_model.predict(new_X_test) )\nprint(\"Test Accuracy: {:.3f}\".format(test_acc) )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T09:28:38.856648Z","iopub.execute_input":"2025-05-22T09:28:38.857034Z","iopub.status.idle":"2025-05-22T09:32:14.276041Z","shell.execute_reply.started":"2025-05-22T09:28:38.857007Z","shell.execute_reply":"2025-05-22T09:32:14.274700Z"}},"outputs":[],"execution_count":null},{"id":"c1956607","cell_type":"markdown","source":"We got a more accurate model.<br>\n\nLet's investigate better the performance using a Confusion Matrix. This is useful to understand better the classes, specifically:\n- Which are the classes that have more instances (important to understand the class that influences more the model)\n- Which are the classes that the model predict better","metadata":{}},{"id":"2c97b3fc","cell_type":"code","source":"from sklearn.metrics import ConfusionMatrixDisplay\n\nConfusionMatrixDisplay.from_estimator(\n    estimator=tuned_model.best_estimator_,\n    X=new_X_test, y=y_test,\n    cmap = 'Blues_r')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T09:32:25.098124Z","iopub.execute_input":"2025-05-22T09:32:25.098663Z","iopub.status.idle":"2025-05-22T09:32:25.493261Z","shell.execute_reply.started":"2025-05-22T09:32:25.098620Z","shell.execute_reply":"2025-05-22T09:32:25.491959Z"}},"outputs":[],"execution_count":null},{"id":"bb87e17b","cell_type":"markdown","source":"We can see that the classes that the model predict better are the one with more instances, this is normal since the goal of the predictor is to predict correctly more instances as possible and not to predict classes in a balanced way. <br>\nIndeed class 0 has the most of the instances, so it has a larger impact on the final measure. <br>\nThe problem we can see is that the predictor classify a lot of instances as class 0, over the 80% of the instances. This number is way too high compared to the baseline accuracy that represent the percentage of instances of class 0 compared to the total (that was 59%). <br>\nWe want to mitigate bias toward the dominant class, to do this we will give to the classes a weight inversely proportional to their frequency.","metadata":{}},{"id":"7421b1c7","cell_type":"code","source":"base_model = RandomForestClassifier(class_weight='balanced') # give weight to the class that are inversely proportional to frequency\nparameters = { 'n_estimators': [50, 100, 200],\n    'max_leaf_nodes': [50, 80, 100],\n    'max_depth': [10, 20, None],\n    'min_samples_split': [2, 5, 10],\n    'bootstrap': [True, False]\n    }\ntuned_model = GridSearchCV(base_model, parameters, cv=5, scoring='accuracy', n_jobs=-1)\ntuned_model.fit(new_X_train, y_train)\nprint (\"Best Score: {:.3f}\".format(tuned_model.best_score_) )\nprint(\"Best Params: \", tuned_model.best_params_)\ntest_acc = accuracy_score(y_true = y_test, y_pred = tuned_model.predict(new_X_test) )\nprint(\"Test Accuracy: {:.3f}\".format(test_acc) )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T09:32:29.976561Z","iopub.execute_input":"2025-05-22T09:32:29.976963Z","iopub.status.idle":"2025-05-22T09:36:19.039601Z","shell.execute_reply.started":"2025-05-22T09:32:29.976932Z","shell.execute_reply":"2025-05-22T09:36:19.038250Z"}},"outputs":[],"execution_count":null},{"id":"b5d5b759","cell_type":"markdown","source":"We can see that the accuracy score is quite the same, but this time we have a more generalized model since it should be able to predict the different labels more equally. <br>\nLet's see, now that we should have a more balanced classificator, how the prediction are distribuited on the labels:","metadata":{}},{"id":"402d864d","cell_type":"code","source":"ConfusionMatrixDisplay.from_estimator(\n    estimator=tuned_model.best_estimator_,\n    X=new_X_test, y=y_test,\n    cmap = 'Blues_r')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T09:46:23.908891Z","iopub.execute_input":"2025-05-22T09:46:23.909306Z","iopub.status.idle":"2025-05-22T09:46:24.297182Z","shell.execute_reply.started":"2025-05-22T09:46:23.909278Z","shell.execute_reply":"2025-05-22T09:46:24.296016Z"}},"outputs":[],"execution_count":null},{"id":"b0723bb7","cell_type":"markdown","source":"The predictions are now more balanced across the classes, indeed the classifier is able to also predict instances that are not from class 0. <br>\nWe can also see that still the instances of with label 3 are difficult to predict since they are few, but at least this time their prediction were wrong but allocated in a more similair class than without the balanced approach. <br>\nFrom now on we will use this last model since it has a better accuracy than the baseline and it's generalized thanks assigning balanced weight to the classes.","metadata":{}},{"id":"46e037c4","cell_type":"markdown","source":"### **Feature Subset Selection**\nRight now we have a model that uses all the features present in the dataset and since they are a lot, this could mean that some of them could be quite irrelevant and what they do is just slow down the speed of the training. <br>\nOur goal is to improve model performance and reduce overfitting, to accomplish this we will now select a subset of the most informative features. <br>\nIn this way we also reduce the risks of random collection and the generalization error of the model.","metadata":{}},{"id":"26d3ea69","cell_type":"markdown","source":"We will start by investigating the importance of the features:","metadata":{}},{"id":"25e873e0","cell_type":"code","source":"feature_names = new_X_train.columns.tolist()\n\nimportant_features = [name for name, importance in zip(feature_names, tuned_model.best_estimator_.feature_importances_) if importance > 0.025]\nprint(important_features)\n\nfig, ax = plt.subplots(figsize=(9, 4))\nax.barh(range(new_X_train.shape[1]), sorted(tuned_model.best_estimator_.feature_importances_)[::-1])\nax.set_title(\"Feature Importances\")\nax.set_yticks(range(new_X_train.shape[1]))\nax.set_yticklabels(np.array(feature_names)[np.argsort(tuned_model.best_estimator_.feature_importances_)[::-1]])\nax.invert_yaxis() \nax.grid()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T09:46:36.598894Z","iopub.execute_input":"2025-05-22T09:46:36.599434Z","iopub.status.idle":"2025-05-22T09:46:38.044072Z","shell.execute_reply.started":"2025-05-22T09:46:36.599391Z","shell.execute_reply":"2025-05-22T09:46:38.042078Z"}},"outputs":[],"execution_count":null},{"id":"b262acbc","cell_type":"markdown","source":"We can see that the most important features are relative to 3 groups of features:\n- Bio-electric Impedance Analysis: measure of key body composition elements, including BMI, fat, muscle, and water content.\n- Physical Measures\n- Sleep Disturbance Scale: scale to categorize sleep disorders in children <br>\n\nFor the selection of the features I will perform a Recursive Elimination in a cross-validation loop. <br>\nIt's a quite slow method but it's more precise than the embedded approach since it removes few features at time and revalidate the model performance each time so it's less affected from dependencies and correlation. <br>\nThis time I will create the classifier with the hyperparameter tuned according to the last model with balanced class weight, since we have already computed the cross-validation.","metadata":{}},{"id":"ec75946f","cell_type":"code","source":"from sklearn.feature_selection import RFECV\nmodel = RandomForestClassifier(class_weight='balanced', bootstrap=True, max_depth=20, max_leaf_nodes=100, min_samples_split=5, n_estimators=200)\nselector = RFECV(model, step=5, cv=5, scoring='accuracy', n_jobs=-1)\nselector.fit(new_X_train, y_train)\nX_train_subset = new_X_train.iloc[:, selector.support_]\nX_test_subset = new_X_test.iloc[:, selector.support_]\nprint(new_X_train.shape[1])\nprint(X_train_subset.shape[1])\nX_train_subset.columns.to_list()\n\nmodel.fit(X_train_subset, y_train)\ntest_acc = accuracy_score(y_true = y_test, y_pred = model.predict(X_test_subset) )\nprint(\"Test Accuracy: {:.3f}\".format(test_acc) )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T09:49:40.116365Z","iopub.execute_input":"2025-05-22T09:49:40.116895Z","iopub.status.idle":"2025-05-22T09:50:56.263540Z","shell.execute_reply.started":"2025-05-22T09:49:40.116834Z","shell.execute_reply":"2025-05-22T09:50:56.262339Z"}},"outputs":[],"execution_count":null},{"id":"a094c14f","cell_type":"markdown","source":"The number of features is a lot lower, indeed we have reduced the features space from 88 to 43. This not only simplyfied the model and reduce the training time, but also mantained the performance since the accuracy score is stable. <br>\nThis suggests us that a good part of the original features were not informative or redundant, so dropping them allowed our model to focus on the more useful features for the prediciton. <br>\nThe Confusion Matrix of this new model composed only by the most important features is:","metadata":{}},{"id":"7f1af7b3","cell_type":"code","source":"ConfusionMatrixDisplay.from_estimator(\n    estimator=model,\n    X=X_test_subset, y=y_test,\n    cmap = 'Blues_r')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T09:55:22.955599Z","iopub.execute_input":"2025-05-22T09:55:22.955982Z","iopub.status.idle":"2025-05-22T09:55:23.340351Z","shell.execute_reply.started":"2025-05-22T09:55:22.955954Z","shell.execute_reply":"2025-05-22T09:55:23.339115Z"}},"outputs":[],"execution_count":null},{"id":"f4a35b9b","cell_type":"markdown","source":"The confusion matrix is quite similiar to the one of the tuned full model, altought we can notice that the accuracy of predicting label 1 has dropped in favor of labels 3 and 4. <br>\nWe can also see which are the features that we have kept:","metadata":{}},{"id":"2c9dee5a","cell_type":"code","source":"subset_feature_names = X_train_subset.columns.tolist()\n\nimportant_features = [name for name, importance in zip(subset_feature_names, model.feature_importances_) if importance > 0.035]\nprint(important_features)\n\nfig, ax = plt.subplots(figsize=(9, 4))\nax.barh(range(X_train_subset.shape[1]), sorted(model.feature_importances_)[::-1])\nax.set_title(\"Feature Importances\")\nax.set_yticks(range(X_train_subset.shape[1]))\nax.set_yticklabels(np.array(subset_feature_names)[np.argsort(model.feature_importances_)[::-1]])\nax.invert_yaxis() \nax.grid()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T10:19:00.621028Z","iopub.execute_input":"2025-05-22T10:19:00.621459Z","iopub.status.idle":"2025-05-22T10:19:01.202383Z","shell.execute_reply.started":"2025-05-22T10:19:00.621430Z","shell.execute_reply":"2025-05-22T10:19:01.201200Z"}},"outputs":[],"execution_count":null},{"id":"08ef5d69","cell_type":"markdown","source":"The first thing we noticed looking at the graph is that the most important features are all numerical, indeed as we could immagine the categorical features weren't that useful for the prediction since they where just informing us in the season the measurement where made. <br>\nThe most important features are the one related to physical measures, sleep disturbance scale and bio-electric impedance analysis, so they don't differ from the one of the full model.","metadata":{}},{"id":"1b5fb321","cell_type":"markdown","source":"### **Evaluation through Kaggle**\nWe want know to evaluate the model through the `test.csv` dataset that represents the one that the Kaggle competition use for the evaluation. <br>\nThe submission file have to be of the format (`id`, `sii`), so for each id we have to predict the correct sii. <br>\nSo we have to transform the test dataset like we did for the training set and then use the definitive model, that is the one with fine tuned parameter and a subset of feature, to predict the labels.","metadata":{}},{"id":"8af10ce8-132b-4551-be60-e70814fe6668","cell_type":"code","source":"# Process the Data Set\ndf_test = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/test.csv\")\nids = df_test['id']\ndel df_test['id']\n\nphysical_measures_df = pd.read_csv('/kaggle/input/nhanes-physical-measures/physical_measures.csv')\n\ndf_test = df_test.merge( physical_measures_df, on=['Basic_Demos-Age', 'Basic_Demos-Sex'], suffixes=('', '_avg'))\ncols = ['Physical-BMI','Physical-Height','Physical-Weight','Physical-Waist_Circumference','Physical-Diastolic_BP','Physical-HeartRate','Physical-Systolic_BP']\ntot_nan_phys = df_test[cols].isna().all(axis=1)\n\nfor col in cols:\n    df_test.loc[tot_nan_phys, col] = df_test.loc[tot_nan_phys, f\"{col}_avg\"]\n    del df_test[f\"{col}_avg\"]\n\nX_test = df_test\n\nis_numerical = np.array([np.issubdtype(dtype, np.number) for dtype in X_test.dtypes])  \nnumerical_idx = np.flatnonzero(is_numerical) \nnew_X_test = X_test.iloc[:, numerical_idx]\n\n\nscaler = StandardScaler()\nimputer = KNNImputer(n_neighbors=3)\n\nscaled_test = scaler.fit_transform(new_X_test)\nX_array = imputer.fit_transform(scaled_test)\nX_array = scaler.inverse_transform(X_array)\nnew_X_test = pd.DataFrame(X_array, columns=new_X_test.columns, index=new_X_test.index)\n\ncategorical_idx = np.flatnonzero(is_numerical==False)\ncategorical_X_test = X_test.iloc[:, categorical_idx]\n\nimputer = SimpleImputer(strategy='most_frequent')\nX_array = imputer.fit_transform(categorical_X_test)\ncategorical_X_test = pd.DataFrame(X_array, columns=categorical_X_test.columns, index=categorical_X_test.index)\n\noh = OneHotEncoder(sparse_output=False)\n\noh.fit(categorical_X_test)\nencoded = oh.transform(categorical_X_test)\n\nfor i, col in enumerate(oh.get_feature_names_out()):\n    new_X_test = new_X_test.copy()\n    new_X_test[col] = encoded[:, i]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T10:33:38.888555Z","iopub.execute_input":"2025-05-22T10:33:38.888909Z","iopub.status.idle":"2025-05-22T10:33:38.971143Z","shell.execute_reply.started":"2025-05-22T10:33:38.888882Z","shell.execute_reply":"2025-05-22T10:33:38.970226Z"}},"outputs":[],"execution_count":null},{"id":"d00e260e-7256-4337-9fcd-89641ef6fe6d","cell_type":"markdown","source":"Since in the test set we have few rows, some of the one-hot encoded features of the train set are not present in the test set, so we have to add them manually and put value 0 in all the instances.","metadata":{}},{"id":"ea92233a-c826-4c9b-aeca-30645aa6948b","cell_type":"code","source":"features_to_add = list(set(feature_names) - set(new_X_test.columns.tolist()))\nfor feature in features_to_add:\n    new_X_test[feature] = 0\n    \n# now we reorder the subset, then we are ready for the prediction task\nnew_X_test = new_X_test[feature_names]\n\nX_test_model = new_X_test.iloc[:, selector.support_]\ny_pred = model.predict(X_test_model)\ny_pred","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T10:33:45.082385Z","iopub.execute_input":"2025-05-22T10:33:45.083433Z","iopub.status.idle":"2025-05-22T10:33:45.115668Z","shell.execute_reply.started":"2025-05-22T10:33:45.083399Z","shell.execute_reply":"2025-05-22T10:33:45.114349Z"}},"outputs":[],"execution_count":null},{"id":"53419281-19e3-471b-9b30-476ef3205208","cell_type":"markdown","source":"Now we are ready to create the csv file for the submission.","metadata":{}},{"id":"0f82cae6-e52e-40f3-8d30-ece90b453cbf","cell_type":"code","source":"submission = pd.DataFrame({\n    'id' : ids,\n    'sii' : y_pred\n})\nsubmission.to_csv('submission.csv',index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T10:33:47.413712Z","iopub.execute_input":"2025-05-22T10:33:47.414094Z","iopub.status.idle":"2025-05-22T10:33:47.421539Z","shell.execute_reply.started":"2025-05-22T10:33:47.414065Z","shell.execute_reply":"2025-05-22T10:33:47.420436Z"}},"outputs":[],"execution_count":null},{"id":"724dee9e-ff8a-47a5-9a61-93ba16e0c86a","cell_type":"markdown","source":"","metadata":{}},{"id":"6df46c77","cell_type":"markdown","source":"### **Investigating Problematic Instances**\nWe are now interested in investigating the instances that are reflected as the most wrong predictions. <br>\nFor most wrong predictions we mean the instances that the model thought were of a class label, but are of another.","metadata":{}},{"id":"2707cbe6","cell_type":"code","source":"y_proba = model.predict_proba(X_test_subset) # for each instances get the probability that it belongs to each class\nconfidence = np.max(y_proba, axis=1) # we get with what probability an instance was mapped to a class label\nresults_df = pd.DataFrame({\n    'true_label': y_test,\n    'pred_label': model.predict(X_test_subset),\n    'confidence': confidence\n})\nresults_df['correct'] = results_df['true_label'] == results_df['pred_label']\nresults_df['index'] = y_test.index\n\nmost_wrong = results_df[~results_df['correct']].sort_values(by='confidence', ascending=False).head(500)\nprint(most_wrong.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T10:34:03.634070Z","iopub.execute_input":"2025-05-22T10:34:03.635085Z","iopub.status.idle":"2025-05-22T10:34:03.713821Z","shell.execute_reply.started":"2025-05-22T10:34:03.635037Z","shell.execute_reply":"2025-05-22T10:34:03.712384Z"}},"outputs":[],"execution_count":null},{"id":"cded241a","cell_type":"markdown","source":"We can see that the instances that were most confident about their results but still predict wrongly are the one that were classified as label 0, but was of label 1.","metadata":{}},{"id":"beb6a618","cell_type":"markdown","source":"We will now see how these instances behave compared to the instances in general. <br>\nTo do this we will take some of the most important features and plot the distribution of the instances to see if they can give us some insight.","metadata":{}},{"id":"a572e32a","cell_type":"code","source":"feature_names = new_X_train.columns.tolist()\n\nimportant_features = [name for name, importance in zip(feature_names, tuned_model.best_estimator_.feature_importances_) if importance > 0.034]\nprint(important_features)\nn = len(important_features)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T10:54:05.313080Z","iopub.execute_input":"2025-05-22T10:54:05.313506Z","iopub.status.idle":"2025-05-22T10:54:05.334454Z","shell.execute_reply.started":"2025-05-22T10:54:05.313471Z","shell.execute_reply":"2025-05-22T10:54:05.333019Z"}},"outputs":[],"execution_count":null},{"id":"2c26b9a0","cell_type":"code","source":"import seaborn as sns\nfig, ax = plt.subplots(1, n, figsize=(20,n))\n\nfor i in range(len(important_features)):\n    sns.kdeplot(X_test_subset[important_features[i]], label='General', ax=ax[i])\n    sns.kdeplot(X_test_subset.loc[most_wrong['index']][important_features[i]], label='Most Wrong', ax=ax[i])\n\n    ax[i].legend()\n    ax[i].set_xlabel(important_features[i])\n    ax[i].set_ylabel('Density')\n    ax[i].set_title(f'Distribution of {important_features[i]}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T10:36:42.185252Z","iopub.execute_input":"2025-05-22T10:36:42.185631Z","iopub.status.idle":"2025-05-22T10:36:43.808097Z","shell.execute_reply.started":"2025-05-22T10:36:42.185604Z","shell.execute_reply":"2025-05-22T10:36:43.806638Z"}},"outputs":[],"execution_count":null},{"id":"ce2a013f","cell_type":"code","source":"most_correct = results_df[results_df['correct']].sort_values(by='confidence', ascending=False).head(500)\nprint(most_correct.head(10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T10:36:54.652590Z","iopub.execute_input":"2025-05-22T10:36:54.653016Z","iopub.status.idle":"2025-05-22T10:36:54.667493Z","shell.execute_reply.started":"2025-05-22T10:36:54.652986Z","shell.execute_reply":"2025-05-22T10:36:54.666121Z"}},"outputs":[],"execution_count":null},{"id":"fec4c43a","cell_type":"code","source":"fig, ax = plt.subplots(1, n, figsize=(20,n))\n\nfor i in range(len(important_features)):\n    sns.kdeplot(X_test_subset[important_features[i]], label='General', ax=ax[i])\n    sns.kdeplot(X_test_subset.loc[most_correct['index']][important_features[i]], label='Most Correct', ax=ax[i])\n\n    ax[i].legend()\n    ax[i].set_xlabel(important_features[i])\n    ax[i].set_ylabel('Density')\n    ax[i].set_title(f'Distribution of {important_features[i]}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-22T10:36:58.232308Z","iopub.execute_input":"2025-05-22T10:36:58.232695Z","iopub.status.idle":"2025-05-22T10:36:59.640894Z","shell.execute_reply.started":"2025-05-22T10:36:58.232668Z","shell.execute_reply":"2025-05-22T10:36:59.639745Z"}},"outputs":[],"execution_count":null}]}