{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport math\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=RuntimeWarning)\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:47.414912Z","iopub.execute_input":"2025-09-23T07:23:47.415113Z","iopub.status.idle":"2025-09-23T07:23:49.428252Z","shell.execute_reply.started":"2025-09-23T07:23:47.415085Z","shell.execute_reply":"2025-09-23T07:23:49.427242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Definitions\n\n#drop_non_numeric(df): Droppa tutte le colonne non numeriche\n#get_fitness_index(df): Droppa le colonne categoriche del FitnessGram Child  e crea una nuova feature con la media di queste\n#impute_negatives(df): Setta a NaN ogni valore negativo\n#fill_BIA(df): Fillo i valori NaN delle colonne BIA con -1 e creo una feature binaria che indica le righe con tutti i valori NaN di queste colonne\n#show_nan(df): Mostra le percentuali di valori NaN per ogni colonna\n#fill_every_nan(df): Imputa tutti i valori NaN delle feature numeriche\n#show_boxplots(df): mostra tutti i boxplots\n#show_boxplots_cols(df, cols): mostra i boxplots di un subset di colonne\n#remove_outliers_iqr(df, cols): rimuove outliers usando il IQR e sostituendo con la mediana","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.429275Z","iopub.execute_input":"2025-09-23T07:23:49.429775Z","iopub.status.idle":"2025-09-23T07:23:49.435863Z","shell.execute_reply.started":"2025-09-23T07:23:49.429749Z","shell.execute_reply":"2025-09-23T07:23:49.434733Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Data preparation**","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntrain_tmp = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\n\nprint(train.shape)\nprint(test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.438572Z","iopub.execute_input":"2025-09-23T07:23:49.438953Z","iopub.status.idle":"2025-09-23T07:23:49.647145Z","shell.execute_reply.started":"2025-09-23T07:23:49.438915Z","shell.execute_reply":"2025-09-23T07:23:49.646237Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Droppo subito queste colonne perchè contengono molti valori NaN nel test set, per valutare il modello su kaggle non ha senso processarle\n#rimuovo anche l'id dal train\ncols = ['id', 'PAQ_A-PAQ_A_Total', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Sec', \n        'Fitness_Endurance-Time_Mins', 'FGC-FGC_GSD', 'FGC-FGC_GSND', 'Physical-Waist_Circumference']\ntrain = train.drop(columns=cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.648018Z","iopub.execute_input":"2025-09-23T07:23:49.648305Z","iopub.status.idle":"2025-09-23T07:23:49.665586Z","shell.execute_reply.started":"2025-09-23T07:23:49.648284Z","shell.execute_reply":"2025-09-23T07:23:49.664535Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#droppo le righe con sii NaN\nprint(train['sii'].isnull().sum())\ntrain = train.dropna(subset=['sii'])\nprint(train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.666597Z","iopub.execute_input":"2025-09-23T07:23:49.666910Z","iopub.status.idle":"2025-09-23T07:23:49.687198Z","shell.execute_reply.started":"2025-09-23T07:23:49.666883Z","shell.execute_reply":"2025-09-23T07:23:49.685934Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#droppo le colonne in train ma non in test\ncols_to_drop = [col for col in train.columns if col not in test.columns and col != 'sii']\ntrain = train.drop(columns=cols_to_drop)\n\nprint(cols_to_drop)\nprint(train.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.688556Z","iopub.execute_input":"2025-09-23T07:23:49.688870Z","iopub.status.idle":"2025-09-23T07:23:49.707337Z","shell.execute_reply.started":"2025-09-23T07:23:49.688841Z","shell.execute_reply":"2025-09-23T07:23:49.706342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.select_dtypes(exclude=['number'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.708305Z","iopub.execute_input":"2025-09-23T07:23:49.708616Z","iopub.status.idle":"2025-09-23T07:23:49.752625Z","shell.execute_reply.started":"2025-09-23T07:23:49.708588Z","shell.execute_reply":"2025-09-23T07:23:49.751602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#droppo le stagioni e l'id\ndef drop_non_numeric(df):\n    cols = df.select_dtypes(exclude=['number']).columns.tolist()\n    \n    if 'id' in cols:\n        cols.remove('id')\n        \n    print(cols)\n    \n    df.drop(columns=cols, inplace=True)\n    print(df.shape)\n\n\ndrop_non_numeric(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.753586Z","iopub.execute_input":"2025-09-23T07:23:49.753931Z","iopub.status.idle":"2025-09-23T07:23:49.762725Z","shell.execute_reply.started":"2025-09-23T07:23:49.753898Z","shell.execute_reply":"2025-09-23T07:23:49.761456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.766196Z","iopub.execute_input":"2025-09-23T07:23:49.766528Z","iopub.status.idle":"2025-09-23T07:23:49.809174Z","shell.execute_reply.started":"2025-09-23T07:23:49.766504Z","shell.execute_reply":"2025-09-23T07:23:49.808190Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#FitnessGram Child (categorical)\ndef get_fitness_index(df):\n\n    cols_FGC = [\n        'FGC-FGC_CU_Zone', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD_Zone',\n        'FGC-FGC_PU_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone',\n        'FGC-FGC_TL_Zone'\n    ]\n\n    # Imputazione con moda\n    for col in cols_FGC:\n        mode = df[col].mode()[0]\n        df[col] = df[col].fillna(mode)\n\n    # Calcolo fitness index\n    df['fitness_index'] = df[cols_FGC].mean(axis=1)    \n    print(df['fitness_index'].describe())\n        \n    df.drop(columns=cols_FGC, inplace=True)\n\n\nget_fitness_index(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.810248Z","iopub.execute_input":"2025-09-23T07:23:49.810754Z","iopub.status.idle":"2025-09-23T07:23:49.841817Z","shell.execute_reply.started":"2025-09-23T07:23:49.810728Z","shell.execute_reply":"2025-09-23T07:23:49.840619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def show_nan(df):\n    perc_nan = df.isnull().mean().sort_values(ascending=False)\n    print(perc_nan[perc_nan > 0])\n\nshow_nan(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.842974Z","iopub.execute_input":"2025-09-23T07:23:49.843323Z","iopub.status.idle":"2025-09-23T07:23:49.853353Z","shell.execute_reply.started":"2025-09-23T07:23:49.843285Z","shell.execute_reply":"2025-09-23T07:23:49.852358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.854416Z","iopub.execute_input":"2025-09-23T07:23:49.854729Z","iopub.status.idle":"2025-09-23T07:23:49.900340Z","shell.execute_reply.started":"2025-09-23T07:23:49.854702Z","shell.execute_reply":"2025-09-23T07:23:49.899401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def impute_negatives(df):\n    cols = df.select_dtypes(include=['number']).columns\n    for col in cols:\n        negatives = df[col] < 0\n        if negatives.any():\n            print(f\"{col}: {negatives.sum()}\")\n            df.loc[negatives, col] = np.nan\n\n        \nimpute_negatives(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.901237Z","iopub.execute_input":"2025-09-23T07:23:49.901552Z","iopub.status.idle":"2025-09-23T07:23:49.919680Z","shell.execute_reply.started":"2025-09-23T07:23:49.901525Z","shell.execute_reply":"2025-09-23T07:23:49.918510Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n#Bio-electric Impedance Analysis\ndef fill_BIA(df):\n    cols_BIA = [\n        'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW',\n        'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', \n        'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW'\n    ]\n    df['BIA_Missing'] = df[cols_BIA].isnull().all(axis=1).astype(int)\n    df[cols_BIA] = df[cols_BIA].fillna(-1) \n\nfill_BIA(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.920793Z","iopub.execute_input":"2025-09-23T07:23:49.921221Z","iopub.status.idle":"2025-09-23T07:23:49.933646Z","shell.execute_reply.started":"2025-09-23T07:23:49.921157Z","shell.execute_reply":"2025-09-23T07:23:49.932570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_nan(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.934582Z","iopub.execute_input":"2025-09-23T07:23:49.934918Z","iopub.status.idle":"2025-09-23T07:23:49.947331Z","shell.execute_reply.started":"2025-09-23T07:23:49.934886Z","shell.execute_reply":"2025-09-23T07:23:49.946054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fill_every_nan(df):\n    for col in df.columns:\n        if df[col].isna().any():\n            if pd.api.types.is_numeric_dtype(df[col]):\n                if df[col].nunique(dropna=True) > 5:\n                    # Colonna numerica continua\n                    median_val = df[col].median()\n                    df[col] = df[col].fillna(median_val)\n                    print(f\"{col}: imputato con mediana ({median_val})\")\n                else:\n                    # Colonna numerica categorica\n                    mode_val = df[col].mode()[0]\n                    df[col] = df[col].fillna(mode_val)\n                    print(f\"{col}: imputato con moda ({mode_val})\")\n\nfill_every_nan(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.948517Z","iopub.execute_input":"2025-09-23T07:23:49.949597Z","iopub.status.idle":"2025-09-23T07:23:49.995708Z","shell.execute_reply.started":"2025-09-23T07:23:49.949562Z","shell.execute_reply":"2025-09-23T07:23:49.994657Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_nan(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:49.996680Z","iopub.execute_input":"2025-09-23T07:23:49.996998Z","iopub.status.idle":"2025-09-23T07:23:50.008384Z","shell.execute_reply.started":"2025-09-23T07:23:49.996970Z","shell.execute_reply":"2025-09-23T07:23:50.007428Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def show_boxplots(df):\n    cols = df.select_dtypes(include=['number']).columns.tolist()\n    \n    n_cols = len(cols)\n    n_rows = math.ceil(n_cols / 6)\n    \n    plt.figure(figsize=(30, 5 * n_rows))\n    \n    for i, col in enumerate(cols, 1):\n        plt.subplot(n_rows, 6, i)\n        sns.boxplot(x=df[col])\n        plt.title(col)\n    \n    plt.tight_layout()\n    plt.show()\n\nshow_boxplots(train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:50.009278Z","iopub.execute_input":"2025-09-23T07:23:50.009587Z","iopub.status.idle":"2025-09-23T07:23:54.509223Z","shell.execute_reply.started":"2025-09-23T07:23:50.009561Z","shell.execute_reply":"2025-09-23T07:23:54.508281Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def show_boxplots_cols(df, cols):\n    \n    n_cols = len(cols)\n    n_rows = math.ceil(n_cols / 6)\n    \n    plt.figure(figsize=(30, 5 * n_rows))\n    \n    for i, col in enumerate(cols, 1):\n        plt.subplot(n_rows, 6, i)\n        sns.boxplot(x=df[col])\n        plt.title(col)\n    \n    plt.tight_layout()\n    plt.show()\n\n\n\ndef remove_outliers_iqr(df, cols):\n    \n    for col in cols:            \n        Q1 = df[col].quantile(0.25)\n        Q3 = df[col].quantile(0.75)\n        IQR = Q3 - Q1\n        lower, upper = Q1 - 1.5 * IQR, Q3 + 1.5 * IQR\n        median = df[col].median()\n\n        train[col] = train[col].where(\n            (train[col] >= lower) & (train[col] <= upper) | train[col].isna(),\n            median\n        )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:54.510224Z","iopub.execute_input":"2025-09-23T07:23:54.510558Z","iopub.status.idle":"2025-09-23T07:23:54.518239Z","shell.execute_reply.started":"2025-09-23T07:23:54.510535Z","shell.execute_reply":"2025-09-23T07:23:54.516859Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols = [\n    'BIA-BIA_BMC', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', \n    'BIA-BIA_Fat', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', \n]\n\nshow_boxplots_cols(train, cols)\n\nremove_outliers_iqr(train, cols)\n\nshow_boxplots_cols(train, cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:54.520034Z","iopub.execute_input":"2025-09-23T07:23:54.520374Z","iopub.status.idle":"2025-09-23T07:23:57.368546Z","shell.execute_reply.started":"2025-09-23T07:23:54.520350Z","shell.execute_reply":"2025-09-23T07:23:57.367499Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Model 1 : Random forest**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report, accuracy_score, confusion_matrix, ConfusionMatrixDisplay\n\nX = train.drop(columns=['sii']) \ny = train['sii']\n\nX_train, X_val, y_train, y_val = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:57.369613Z","iopub.execute_input":"2025-09-23T07:23:57.369943Z","iopub.status.idle":"2025-09-23T07:23:58.003484Z","shell.execute_reply.started":"2025-09-23T07:23:57.369920Z","shell.execute_reply":"2025-09-23T07:23:58.002526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# GridSearch\nparam_grid = {\n    'n_estimators': [50, 100, 200],\n    'max_depth': [None, 10, 20],\n    'min_samples_split': [2, 5],\n    'min_samples_leaf': [1, 2],\n    'max_features': ['sqrt', 'log2', None]\n}\n\nrf = RandomForestClassifier(random_state=42)\n\ngrid_search = GridSearchCV(\n    estimator=rf,\n    param_grid=param_grid,\n    scoring='accuracy',\n    cv=3,\n    n_jobs=-1,\n    verbose=2\n)\n\ngrid_search.fit(X_train, y_train)\n\nprint(f\"Best parameters: {grid_search.best_params_}\")\nprint(f\"Best CV accuracy: {grid_search.best_score_:.4f}\")\n\nbest_rf = grid_search.best_estimator_\ny_val_pred = best_rf.predict(X_val)\n\nprint(\"\\n\\nClassification report:\")\nprint(classification_report(y_val, y_val_pred, zero_division=0))\nprint(\"\\nAccuracy:\", accuracy_score(y_val, y_val_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:23:58.004474Z","iopub.execute_input":"2025-09-23T07:23:58.004826Z","iopub.status.idle":"2025-09-23T07:26:35.296595Z","shell.execute_reply.started":"2025-09-23T07:23:58.004797Z","shell.execute_reply":"2025-09-23T07:26:35.295662Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature importance\nimportances = best_rf.feature_importances_\nfeatures = X_train.columns\n\nfi_df = pd.DataFrame({\n    'Feature': features,\n    'Importance': importances\n}).sort_values(by='Importance', ascending=False)\n\nprint(\"\\nTop 20 feature più importanti:\")\nprint(fi_df.head(20))\n\nplt.figure(figsize=(20, 6))\nsns.barplot(x='Importance', y='Feature', data=fi_df.head(10), palette='viridis')\nplt.title(\"Feature Importances - Random Forest\")\nplt.tight_layout()\nplt.savefig(\"random_forest_feature_importance\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:26:35.297495Z","iopub.execute_input":"2025-09-23T07:26:35.297736Z","iopub.status.idle":"2025-09-23T07:26:35.755640Z","shell.execute_reply.started":"2025-09-23T07:26:35.297718Z","shell.execute_reply":"2025-09-23T07:26:35.754595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_results = X_val.copy()\nval_results['true_label'] = y_val\nval_results['predicted'] = y_val_pred\nval_results['correct'] = val_results['true_label'] == val_results['predicted']\n\nerror_counts = val_results[val_results['correct'] == False]['true_label'].value_counts()\ncorrect_counts = val_results[val_results['correct'] == True]['true_label'].value_counts()\n\nprint(\"\\nWrong:\")\nprint(error_counts)\nprint(\"\\nCorrect:\")\nprint(correct_counts)\n\ncombined = pd.DataFrame({\n    'Correct': correct_counts,\n    'Wrong': error_counts\n}).fillna(0)\n\ncombined.plot(kind='bar', figsize=(10, 5), title=\"Correct vs Wrong Predictions\")\nplt.ylabel(\"Numero di istanze\")\nplt.xticks(rotation=45)\nplt.grid(axis='y')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:26:35.756586Z","iopub.execute_input":"2025-09-23T07:26:35.756909Z","iopub.status.idle":"2025-09-23T07:26:36.068492Z","shell.execute_reply.started":"2025-09-23T07:26:35.756881Z","shell.execute_reply":"2025-09-23T07:26:36.067549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Confusion Matrix\ncm = confusion_matrix(y_val, y_val_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=best_rf.classes_)\n\nplt.figure(figsize=(6, 6))\ndisp.plot(cmap='Blues', values_format='d')\nplt.title(\"Confusion Matrix - Random Forest\")\nplt.grid(False)\nplt.savefig(\"random_forest_confusion_matrix\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:26:36.069544Z","iopub.execute_input":"2025-09-23T07:26:36.069860Z","iopub.status.idle":"2025-09-23T07:26:36.471985Z","shell.execute_reply.started":"2025-09-23T07:26:36.069833Z","shell.execute_reply":"2025-09-23T07:26:36.471051Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Model 2 : XGBoost**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.metrics import classification_report, accuracy_score, confusion_matrix, ConfusionMatrixDisplay\nfrom xgboost import XGBClassifier\n\nX = train.drop(columns=['sii'])\ny = train['sii']\n\nX_train, X_val, y_train, y_val = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:26:36.473177Z","iopub.execute_input":"2025-09-23T07:26:36.473566Z","iopub.status.idle":"2025-09-23T07:26:36.912045Z","shell.execute_reply.started":"2025-09-23T07:26:36.473537Z","shell.execute_reply":"2025-09-23T07:26:36.910655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"param_grid = {\n    'n_estimators': [50, 100, 200],\n    'max_depth': [3, 6, 10], \n    'learning_rate': [0.01, 0.1, 0.2],\n    'subsample': [0.7, 1.0],\n    'colsample_bytree': [0.7, 1.0]\n}\n\nxgb = XGBClassifier(random_state=42, use_label_encoder=False, eval_metric='mlogloss')\n\ngrid_search = GridSearchCV(\n    estimator=xgb,\n    param_grid=param_grid,\n    scoring='accuracy',\n    cv=3,\n    n_jobs=-1,\n    verbose=2\n)\n\ngrid_search.fit(X_train, y_train)\n\nprint(f\"Best parameters: {grid_search.best_params_}\")\nprint(f\"Best CV accuracy: {grid_search.best_score_:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:26:36.916357Z","iopub.execute_input":"2025-09-23T07:26:36.917502Z","iopub.status.idle":"2025-09-23T07:30:34.822699Z","shell.execute_reply.started":"2025-09-23T07:26:36.917347Z","shell.execute_reply":"2025-09-23T07:30:34.821327Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_xgb = grid_search.best_estimator_\ny_val_pred = best_xgb.predict(X_val)\n\nprint(\"\\n\\nClassification report:\")\nprint(classification_report(y_val, y_val_pred, zero_division=0))\nprint(\"\\nAccuracy:\", accuracy_score(y_val, y_val_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:34.823310Z","iopub.execute_input":"2025-09-23T07:30:34.823546Z","iopub.status.idle":"2025-09-23T07:30:34.848350Z","shell.execute_reply.started":"2025-09-23T07:30:34.823527Z","shell.execute_reply":"2025-09-23T07:30:34.847579Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"importances = best_xgb.feature_importances_\nfeatures = X_train.columns\n\nfi_df = pd.DataFrame({\n    'Feature': features,\n    'Importance': importances\n}).sort_values(by='Importance', ascending=False)\n\nprint(\"\\nTop 20 feature più importanti:\")\nprint(fi_df.head(20))\n\nplt.figure(figsize=(20, 6))\nsns.barplot(x='Importance', y='Feature', data=fi_df.head(10), palette='viridis')\nplt.title(\"Feature Importances - XGBoost\")\nplt.tight_layout()\nplt.savefig(\"xgboost_feature_importance.png\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:34.851126Z","iopub.execute_input":"2025-09-23T07:30:34.852588Z","iopub.status.idle":"2025-09-23T07:30:35.297583Z","shell.execute_reply.started":"2025-09-23T07:30:34.852557Z","shell.execute_reply":"2025-09-23T07:30:35.296620Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_results = X_val.copy()\nval_results['true_label'] = y_val\nval_results['predicted'] = y_val_pred\nval_results['correct'] = val_results['true_label'] == val_results['predicted']\n\nerror_counts = val_results[val_results['correct'] == False]['true_label'].value_counts()\ncorrect_counts = val_results[val_results['correct'] == True]['true_label'].value_counts()\n\nprint(\"\\nWrong:\")\nprint(error_counts)\nprint(\"\\nCorrect:\")\nprint(correct_counts)\n\ncombined = pd.DataFrame({\n    'Correct': correct_counts,\n    'Wrong': error_counts\n}).fillna(0)\n\ncombined.plot(kind='bar', figsize=(10, 5), title=\"Correct vs Wrong Predictions\")\nplt.ylabel(\"Numero di istanze\")\nplt.xticks(rotation=45)\nplt.grid(axis='y')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:35.299121Z","iopub.execute_input":"2025-09-23T07:30:35.299433Z","iopub.status.idle":"2025-09-23T07:30:35.529609Z","shell.execute_reply.started":"2025-09-23T07:30:35.299410Z","shell.execute_reply":"2025-09-23T07:30:35.528677Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cm = confusion_matrix(y_val, y_val_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=best_xgb.classes_)\n\nplt.figure(figsize=(6, 6))\ndisp.plot(cmap='Blues', values_format='d')\nplt.title(\"Confusion Matrix - XGBoost\")\nplt.grid(False)\nplt.savefig(\"xgboost_confusion_matrix.png\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:35.530749Z","iopub.execute_input":"2025-09-23T07:30:35.531067Z","iopub.status.idle":"2025-09-23T07:30:35.854230Z","shell.execute_reply.started":"2025-09-23T07:30:35.531040Z","shell.execute_reply":"2025-09-23T07:30:35.853382Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Submission**","metadata":{}},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ntest.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:35.855106Z","iopub.execute_input":"2025-09-23T07:30:35.855448Z","iopub.status.idle":"2025-09-23T07:30:35.867069Z","shell.execute_reply.started":"2025-09-23T07:30:35.855425Z","shell.execute_reply":"2025-09-23T07:30:35.866333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"drop_non_numeric(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:35.868057Z","iopub.execute_input":"2025-09-23T07:30:35.868423Z","iopub.status.idle":"2025-09-23T07:30:35.881069Z","shell.execute_reply.started":"2025-09-23T07:30:35.868399Z","shell.execute_reply":"2025-09-23T07:30:35.880267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"get_fitness_index(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:35.881994Z","iopub.execute_input":"2025-09-23T07:30:35.882325Z","iopub.status.idle":"2025-09-23T07:30:35.906325Z","shell.execute_reply.started":"2025-09-23T07:30:35.882299Z","shell.execute_reply":"2025-09-23T07:30:35.905351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#droppo le colonne che non ci sono nel train\ncols = [col for col in test.columns if col not in train.columns and col != 'id']\ntest = test.drop(columns=cols)\ntest.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:35.907163Z","iopub.execute_input":"2025-09-23T07:30:35.907502Z","iopub.status.idle":"2025-09-23T07:30:35.915406Z","shell.execute_reply.started":"2025-09-23T07:30:35.907479Z","shell.execute_reply":"2025-09-23T07:30:35.914361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"impute_negatives(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:35.916689Z","iopub.execute_input":"2025-09-23T07:30:35.916943Z","iopub.status.idle":"2025-09-23T07:30:35.945167Z","shell.execute_reply.started":"2025-09-23T07:30:35.916921Z","shell.execute_reply":"2025-09-23T07:30:35.944119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_nan(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:35.946443Z","iopub.execute_input":"2025-09-23T07:30:35.946771Z","iopub.status.idle":"2025-09-23T07:30:35.955094Z","shell.execute_reply.started":"2025-09-23T07:30:35.946737Z","shell.execute_reply":"2025-09-23T07:30:35.954251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fill_BIA(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:35.956153Z","iopub.execute_input":"2025-09-23T07:30:35.956806Z","iopub.status.idle":"2025-09-23T07:30:35.977007Z","shell.execute_reply.started":"2025-09-23T07:30:35.956775Z","shell.execute_reply":"2025-09-23T07:30:35.976182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fill_every_nan(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:35.978003Z","iopub.execute_input":"2025-09-23T07:30:35.978346Z","iopub.status.idle":"2025-09-23T07:30:36.005563Z","shell.execute_reply.started":"2025-09-23T07:30:35.978316Z","shell.execute_reply":"2025-09-23T07:30:36.004640Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"show_boxplots(test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:36.006538Z","iopub.execute_input":"2025-09-23T07:30:36.006892Z","iopub.status.idle":"2025-09-23T07:30:40.254670Z","shell.execute_reply.started":"2025-09-23T07:30:36.006853Z","shell.execute_reply":"2025-09-23T07:30:40.253285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cols = [\n    #'CGAS-CGAS_Score'\n]\n\nshow_boxplots_cols(train, cols)\n\nremove_outliers_iqr(train, cols)\n\nshow_boxplots_cols(train, cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:40.256095Z","iopub.execute_input":"2025-09-23T07:30:40.256437Z","iopub.status.idle":"2025-09-23T07:30:40.267392Z","shell.execute_reply.started":"2025-09-23T07:30:40.256411Z","shell.execute_reply":"2025-09-23T07:30:40.266473Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"#Lancio XGBoost\nX_test = test.drop(columns=['id'])\n\ny_test_pred = best_xgb.predict(X_test)\n\n\nresults = pd.DataFrame({\n    'id': test['id'],\n    'sii': y_test_pred\n})\n\nresults.to_csv('submission.csv', index=False)\n\nresults","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-23T07:30:40.269024Z","iopub.execute_input":"2025-09-23T07:30:40.269382Z","iopub.status.idle":"2025-09-23T07:30:40.306631Z","shell.execute_reply.started":"2025-09-23T07:30:40.269360Z","shell.execute_reply":"2025-09-23T07:30:40.305136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}