{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport time\nimport seaborn as sns\nfrom sklearn.cluster import KMeans\nfrom sklearn.preprocessing import OrdinalEncoder, StandardScaler\nfrom sklearn.model_selection import train_test_split, GridSearchCV, cross_val_score, KFold\nfrom sklearn.svm import SVC\nfrom sklearn.linear_model import LogisticRegressionCV\nfrom sklearn.ensemble import GradientBoostingClassifier, RandomForestClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nfrom sklearn.decomposition import PCA\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\nfrom scipy import stats","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:06.026230Z","iopub.execute_input":"2024-12-02T02:00:06.026951Z","iopub.status.idle":"2024-12-02T02:00:06.034481Z","shell.execute_reply.started":"2024-12-02T02:00:06.026900Z","shell.execute_reply":"2024-12-02T02:00:06.033331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:06.035979Z","iopub.execute_input":"2024-12-02T02:00:06.036490Z","iopub.status.idle":"2024-12-02T02:00:06.107501Z","shell.execute_reply.started":"2024-12-02T02:00:06.036436Z","shell.execute_reply":"2024-12-02T02:00:06.106347Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Exploratory Analysis","metadata":{}},{"cell_type":"markdown","source":"### Age distribution","metadata":{}},{"cell_type":"code","source":"sns.set(style=\"whitegrid\")\n\nplt.figure(figsize=(8, 6))\nsns.histplot(train['Basic_Demos-Age'], bins=20, kde=True, color='skyblue')\nplt.title('Age Distribution')\nplt.xlabel('Age')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:06.108785Z","iopub.execute_input":"2024-12-02T02:00:06.109160Z","iopub.status.idle":"2024-12-02T02:00:06.500586Z","shell.execute_reply.started":"2024-12-02T02:00:06.109123Z","shell.execute_reply":"2024-12-02T02:00:06.499329Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Gender relation","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(6, 6))\n\nsex_counts = train['Basic_Demos-Sex'].value_counts()\n\nsex_labels = ['Feminine', 'Masculine']\n\nplt.pie(sex_counts, labels=sex_labels, autopct='%1.1f%%', colors=['#ff9999', '#66b3ff'], startangle=140)\n\nplt.title('Sex Distribution')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:06.504263Z","iopub.execute_input":"2024-12-02T02:00:06.504787Z","iopub.status.idle":"2024-12-02T02:00:06.624161Z","shell.execute_reply.started":"2024-12-02T02:00:06.504732Z","shell.execute_reply":"2024-12-02T02:00:06.622749Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### BMI Distribution","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\n\nsns.histplot(train['Physical-BMI'].dropna(), bins=30, kde=True, color='green')\n\nplt.title('BMI Distribution')\n\nplt.xlabel('BMI')\n\nplt.ylabel('Frequency')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:06.625744Z","iopub.execute_input":"2024-12-02T02:00:06.626303Z","iopub.status.idle":"2024-12-02T02:00:07.110487Z","shell.execute_reply.started":"2024-12-02T02:00:06.626238Z","shell.execute_reply":"2024-12-02T02:00:07.109415Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Null values per attribute","metadata":{}},{"cell_type":"code","source":"null_counts = train.isnull().sum()\n\nnull_counts = null_counts[null_counts > 0].sort_values(ascending=False)\n\nplt.figure(figsize=(12, 8))\n\nsns.barplot(x=null_counts.values, y=null_counts.index, palette=\"viridis\")\n\nplt.title('Null Values per Column')\n\nplt.xlabel('Number of Null Values')\n\nplt.ylabel('Columns')\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:07.112412Z","iopub.execute_input":"2024-12-02T02:00:07.112885Z","iopub.status.idle":"2024-12-02T02:00:08.166160Z","shell.execute_reply.started":"2024-12-02T02:00:07.112834Z","shell.execute_reply":"2024-12-02T02:00:08.165094Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data cleaning and feature engineering","metadata":{}},{"cell_type":"markdown","source":"First thing we notice is that the columns in the training dataset and testing dataset are different, so we keep only the ones that are in both.","metadata":{}},{"cell_type":"code","source":"common_columns = train.columns.intersection(test.columns)\n\ntrain = train[common_columns.to_list() + ['sii']]\ntest = test[common_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:08.167757Z","iopub.execute_input":"2024-12-02T02:00:08.168231Z","iopub.status.idle":"2024-12-02T02:00:08.177331Z","shell.execute_reply.started":"2024-12-02T02:00:08.168180Z","shell.execute_reply":"2024-12-02T02:00:08.176319Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We're imputing the attributes with less than 50% missing data, the columns with more than 50% missing data will be removed","metadata":{}},{"cell_type":"code","source":"total_rows = train.shape[0]\n\nto_eliminate = []\n\nfor col in train.columns:\n    if col == \"sii\":\n        continue\n    missing_count = train[col].isnull().sum()\n    if missing_count / total_rows > 0.5:\n        to_eliminate.append(col)\n    else:\n        # Imputar con valor promedio si es numerico, moda si es categorico\n        if pd.api.types.is_integer_dtype(train[col]):\n            train[col] = train[col].fillna(train[col].mode())\n            test[col] = test[col].fillna(test[col].mode())\n        elif pd.api.types.is_float_dtype(train[col]):\n            train[col] = train[col].fillna(train[col].mean())\n            test[col] = test[col].fillna(test[col].mean())\n#print(to_eliminate)\ntrain.drop(to_eliminate, axis=1, inplace=True)\ntest.drop(to_eliminate, axis=1, inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:08.178600Z","iopub.execute_input":"2024-12-02T02:00:08.178977Z","iopub.status.idle":"2024-12-02T02:00:08.238961Z","shell.execute_reply.started":"2024-12-02T02:00:08.178942Z","shell.execute_reply":"2024-12-02T02:00:08.237849Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now, we encode the categorical variables with OrdinalEncoder and the numerical variables with StandardScaler","metadata":{}},{"cell_type":"code","source":"for col in train.columns:\n    if \"Season\" in col:\n        train.loc[:, col] = train[col].fillna(\"Missing\")\n        test.loc[:, col] = test[col].fillna(\"Missing\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:08.240419Z","iopub.execute_input":"2024-12-02T02:00:08.240771Z","iopub.status.idle":"2024-12-02T02:00:08.256414Z","shell.execute_reply.started":"2024-12-02T02:00:08.240716Z","shell.execute_reply":"2024-12-02T02:00:08.255280Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ordinal_encoder = OrdinalEncoder(categories=[[\"Missing\",\"Spring\", \"Summer\", \"Fall\", \"Winter\"]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:08.257842Z","iopub.execute_input":"2024-12-02T02:00:08.258212Z","iopub.status.idle":"2024-12-02T02:00:08.270561Z","shell.execute_reply.started":"2024-12-02T02:00:08.258176Z","shell.execute_reply":"2024-12-02T02:00:08.269569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"season_cols = []\nfor col in list(train.columns):\n    if \"Season\" in col:\n        train.loc[:, col] = ordinal_encoder.fit_transform(train[[col]])\n        test.loc[:, col] = ordinal_encoder.transform(test[[col]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:08.272057Z","iopub.execute_input":"2024-12-02T02:00:08.272438Z","iopub.status.idle":"2024-12-02T02:00:08.310670Z","shell.execute_reply.started":"2024-12-02T02:00:08.272404Z","shell.execute_reply":"2024-12-02T02:00:08.309529Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler()\nfor col in train.columns:\n    if col == \"sii\":\n        continue\n    if pd.api.types.is_numeric_dtype(train[col]):\n        train[[col]] = scaler.fit_transform(train[[col]])\n        test[[col]] = scaler.transform(test[[col]])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:08.312313Z","iopub.execute_input":"2024-12-02T02:00:08.312665Z","iopub.status.idle":"2024-12-02T02:00:08.489525Z","shell.execute_reply.started":"2024-12-02T02:00:08.312631Z","shell.execute_reply":"2024-12-02T02:00:08.488182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:08.493346Z","iopub.execute_input":"2024-12-02T02:00:08.494181Z","iopub.status.idle":"2024-12-02T02:00:08.522878Z","shell.execute_reply.started":"2024-12-02T02:00:08.494137Z","shell.execute_reply":"2024-12-02T02:00:08.521776Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## PCA\n\nPCA can be used for visualization of the dataset in a 2D or 3D space, but it can also be used as a tool to manipulate data.","metadata":{}},{"cell_type":"markdown","source":"### Three dimension visualization","metadata":{}},{"cell_type":"code","source":"pca = PCA(n_components=3)\nprincipal_components = pca.fit_transform(train.drop([\"id\", \"sii\"], axis=1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:08.524493Z","iopub.execute_input":"2024-12-02T02:00:08.524856Z","iopub.status.idle":"2024-12-02T02:00:08.631987Z","shell.execute_reply.started":"2024-12-02T02:00:08.524822Z","shell.execute_reply":"2024-12-02T02:00:08.630933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"principal_df = pd.DataFrame(data = principal_components\n             , columns = ['principal component 1', 'principal component 2', 'principal component 3'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:08.633815Z","iopub.execute_input":"2024-12-02T02:00:08.634461Z","iopub.status.idle":"2024-12-02T02:00:08.649427Z","shell.execute_reply.started":"2024-12-02T02:00:08.634404Z","shell.execute_reply":"2024-12-02T02:00:08.645387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df = pd.concat([principal_df, train[['sii']]], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:08.651248Z","iopub.execute_input":"2024-12-02T02:00:08.651668Z","iopub.status.idle":"2024-12-02T02:00:08.671218Z","shell.execute_reply.started":"2024-12-02T02:00:08.651625Z","shell.execute_reply":"2024-12-02T02:00:08.670100Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = plt.figure(figsize=(10, 10))\nax = fig.add_subplot(111, projection='3d')  # Configura la proyección en 3D\nax.set_xlabel('Principal Component 1', fontsize=15)\nax.set_ylabel('Principal Component 2', fontsize=15)\nax.set_zlabel('Principal Component 3', fontsize=15)\nax.set_title('3 Component PCA', fontsize=20)\n\n# Asumiendo que tienes tres componentes principales en tu DataFrame\ntargets = [0, 1, 2, 3]\ncolors = ['r', 'g', 'b', 'y']  # Se añadió un color más para el cuarto target\n\nfor target, color in zip(targets, colors):\n    indicesToKeep = final_df['sii'] == target\n    ax.scatter(final_df.loc[indicesToKeep, 'principal component 1'],\n               final_df.loc[indicesToKeep, 'principal component 2'],\n               final_df.loc[indicesToKeep, 'principal component 3'],\n               c=color,\n               s=50,\n               label=f'Target {target}')\n\nax.legend()\nax.grid()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:08.672745Z","iopub.execute_input":"2024-12-02T02:00:08.673291Z","iopub.status.idle":"2024-12-02T02:00:09.279966Z","shell.execute_reply.started":"2024-12-02T02:00:08.673244Z","shell.execute_reply":"2024-12-02T02:00:09.278209Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Two dimension visualization","metadata":{}},{"cell_type":"code","source":"pca = PCA(n_components=2)\nprincipal_components = pca.fit_transform(train.drop([\"id\", \"sii\"], axis=1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:09.281837Z","iopub.execute_input":"2024-12-02T02:00:09.282514Z","iopub.status.idle":"2024-12-02T02:00:09.387526Z","shell.execute_reply.started":"2024-12-02T02:00:09.282459Z","shell.execute_reply":"2024-12-02T02:00:09.385851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"principal_df = pd.DataFrame(data = principal_components\n             , columns = ['principal component 1', 'principal component 2'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:09.388905Z","iopub.execute_input":"2024-12-02T02:00:09.389313Z","iopub.status.idle":"2024-12-02T02:00:09.403294Z","shell.execute_reply.started":"2024-12-02T02:00:09.389269Z","shell.execute_reply":"2024-12-02T02:00:09.401378Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"final_df = pd.concat([principal_df, train[['sii']]], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:09.404788Z","iopub.execute_input":"2024-12-02T02:00:09.408975Z","iopub.status.idle":"2024-12-02T02:00:09.422442Z","shell.execute_reply.started":"2024-12-02T02:00:09.408890Z","shell.execute_reply":"2024-12-02T02:00:09.421268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = plt.figure(figsize = (8,8))\nax = fig.add_subplot(1,1,1) \nax.set_xlabel('Principal Component 1', fontsize = 15)\nax.set_ylabel('Principal Component 2', fontsize = 15)\nax.set_title('2 component PCA', fontsize = 20)\n\ntargets = [0, 1, 2, 3]\ncolors = ['r', 'g', 'b', 'y']\nfor target, color in zip(targets,colors):\n    indicesToKeep = final_df['sii'] == target\n    ax.scatter(final_df.loc[indicesToKeep, 'principal component 1']\n               , final_df.loc[indicesToKeep, 'principal component 2']\n               , c = color\n               , s = 50)\nax.legend(targets)\nax.grid()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:09.423866Z","iopub.execute_input":"2024-12-02T02:00:09.424373Z","iopub.status.idle":"2024-12-02T02:00:10.090587Z","shell.execute_reply.started":"2024-12-02T02:00:09.424311Z","shell.execute_reply":"2024-12-02T02:00:10.089123Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**As we see, PCA for visualization is not really helpful in this dataset as it does not show any clear patterns among the data**","metadata":{}},{"cell_type":"markdown","source":"## Unsupervised learning\n\nSince this is a classification problem, and there are some rows that have a null target variable, on our first approach we're going to fill that missing data with the use of KMeans.","metadata":{}},{"cell_type":"code","source":"kmeans = KMeans(n_clusters=4, n_init=\"auto\", random_state=42).fit(train.drop([\"id\", \"sii\"], axis=1).dropna())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:10.092576Z","iopub.execute_input":"2024-12-02T02:00:10.092951Z","iopub.status.idle":"2024-12-02T02:00:10.264626Z","shell.execute_reply.started":"2024-12-02T02:00:10.092915Z","shell.execute_reply":"2024-12-02T02:00:10.263610Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = kmeans.predict(train[train.isnull().any(axis=1)].drop([\"id\", \"sii\"], axis=1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:10.265679Z","iopub.execute_input":"2024-12-02T02:00:10.266031Z","iopub.status.idle":"2024-12-02T02:00:10.293308Z","shell.execute_reply.started":"2024-12-02T02:00:10.265978Z","shell.execute_reply":"2024-12-02T02:00:10.292410Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Crear una copia para evitar alterar el dataset original\ndata_filled = train.copy()\n\n# Filtrar filas con valores faltantes en 'sii'\nmissing_sii_indices = data_filled['sii'].isnull()\n\n# Preparar los datos eliminando columnas irrelevantes para KMeans\nfeatures_for_prediction = data_filled.loc[missing_sii_indices].drop([\"id\", \"sii\"], axis=1)\n\n# Predecir los valores faltantes\npredictions = kmeans.predict(features_for_prediction)\n\n# Asignar las predicciones a los valores faltantes en 'sii'\ndata_filled.loc[missing_sii_indices, 'sii'] = predictions\n\n# Verificar los cambios\nprint(data_filled['sii'].isnull().sum())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:10.294888Z","iopub.execute_input":"2024-12-02T02:00:10.295251Z","iopub.status.idle":"2024-12-02T02:00:10.330481Z","shell.execute_reply.started":"2024-12-02T02:00:10.295215Z","shell.execute_reply":"2024-12-02T02:00:10.327996Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_filled.isnull().sum().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:10.334308Z","iopub.execute_input":"2024-12-02T02:00:10.337888Z","iopub.status.idle":"2024-12-02T02:00:10.348232Z","shell.execute_reply.started":"2024-12-02T02:00:10.337833Z","shell.execute_reply":"2024-12-02T02:00:10.347252Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_filled.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:10.349788Z","iopub.execute_input":"2024-12-02T02:00:10.350280Z","iopub.status.idle":"2024-12-02T02:00:10.450754Z","shell.execute_reply.started":"2024-12-02T02:00:10.350228Z","shell.execute_reply":"2024-12-02T02:00:10.449489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Seleccionar solo las columnas numéricas\nnumerical_features = data_filled.drop(\"id\", axis=1).columns\n\n# Calcular correlación de cada columna numérica con 'sii'\ncorrelations = data_filled[numerical_features].corrwith(data_filled['sii'])\n\n# Ordenar las correlaciones de mayor a menor\ncorrelations = correlations.sort_values(ascending=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:10.452365Z","iopub.execute_input":"2024-12-02T02:00:10.452808Z","iopub.status.idle":"2024-12-02T02:00:10.483638Z","shell.execute_reply.started":"2024-12-02T02:00:10.452754Z","shell.execute_reply":"2024-12-02T02:00:10.482718Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Supongamos que \"correlations\" ya está calculado\n# Ordenar las correlaciones de mayor a menor\ncorrelations_sorted = correlations.sort_values(ascending=False)\n\n# Crear la gráfica\nplt.figure(figsize=(10, 6))\ncorrelations_sorted.plot(kind=\"bar\", color=\"skyblue\")\n\n# Configuración del gráfico\nplt.title(\"Correlación de las variables con 'sii'\", fontsize=16)\nplt.xlabel(\"Variables\", fontsize=12)\nplt.ylabel(\"Correlación\", fontsize=12)\nplt.xticks(rotation=45, ha='right')\nplt.tight_layout()\n\n# Mostrar el gráfico\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:10.484909Z","iopub.execute_input":"2024-12-02T02:00:10.485249Z","iopub.status.idle":"2024-12-02T02:00:11.530846Z","shell.execute_reply.started":"2024-12-02T02:00:10.485216Z","shell.execute_reply":"2024-12-02T02:00:11.529714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Definir el umbral mínimo de correlación (por ejemplo, 0.1)\nthreshold = 0.1\n\n# Filtrar las columnas que cumplen con el umbral\ncolumns_to_keep = correlations[correlations.abs() >= threshold].index\n\n# Crear un nuevo DataFrame con las columnas seleccionadas\nfiltered_train = data_filled[columns_to_keep]\ncolumns_to_keep_for_test = columns_to_keep.difference([\"sii\"])\n\n# Crear un nuevo DataFrame para test\nfiltered_test = test[columns_to_keep_for_test]\n\n# Mostrar las columnas restantes\nprint(f\"Selected columns (correlation >= {threshold}):\")\nprint(filtered_train.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:11.532317Z","iopub.execute_input":"2024-12-02T02:00:11.532657Z","iopub.status.idle":"2024-12-02T02:00:11.543526Z","shell.execute_reply.started":"2024-12-02T02:00:11.532623Z","shell.execute_reply":"2024-12-02T02:00:11.542383Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# X, y splits","metadata":{}},{"cell_type":"code","source":"X = data_filled.drop([\"sii\", \"id\"], axis=1)\ny = data_filled[\"sii\"]\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, stratify=y, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:20.028729Z","iopub.execute_input":"2024-12-02T02:00:20.029687Z","iopub.status.idle":"2024-12-02T02:00:20.042390Z","shell.execute_reply.started":"2024-12-02T02:00:20.029635Z","shell.execute_reply":"2024-12-02T02:00:20.041167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:20.563026Z","iopub.execute_input":"2024-12-02T02:00:20.563404Z","iopub.status.idle":"2024-12-02T02:00:20.590656Z","shell.execute_reply.started":"2024-12-02T02:00:20.563366Z","shell.execute_reply":"2024-12-02T02:00:20.589347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_pca_results(classifier, X_train, y_train, X_test, y_test):\n    variance_retained = [1, 0.99, 0.95, 0.9, 0.85]\n    pca_results = pd.DataFrame({\"Variance Retained\": variance_retained})\n    array_components = [X_train.shape[1]]\n    start_time = time.time()\n    classifier.fit(X_train, y_train)\n    end_time = time.time()\n    array_scores = [classifier.score(X_test, y_test)]\n    array_times = [f\"{(end_time - start_time):.5f}\"]\n    for var_retained in variance_retained[1:]:\n        pca = PCA(var_retained, random_state=42)\n        pca.fit(X_train)\n        temp_train = pca.transform(X_train)\n        temp_test = pca.transform(X_test)\n        start_time = time.time()\n        classifier.fit(temp_train, y_train)\n        end_time = time.time()\n        score = classifier.score(temp_test, y_test)\n        array_components.append(pca.n_components_)\n        array_scores.append(score)\n        array_times.append(f\"{(end_time - start_time):.5f}\")\n    pca_results[\"Number of components\"] = array_components\n    pca_results[\"Time to train\"] = array_times\n    pca_results[\"Accuracy\"] = array_scores\n    return pca_results","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:22.654487Z","iopub.execute_input":"2024-12-02T02:00:22.655245Z","iopub.status.idle":"2024-12-02T02:00:22.663936Z","shell.execute_reply.started":"2024-12-02T02:00:22.655193Z","shell.execute_reply":"2024-12-02T02:00:22.662660Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.shape, X_test.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:23.372417Z","iopub.execute_input":"2024-12-02T02:00:23.372803Z","iopub.status.idle":"2024-12-02T02:00:23.380135Z","shell.execute_reply.started":"2024-12-02T02:00:23.372769Z","shell.execute_reply":"2024-12-02T02:00:23.379047Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SVC Model\n\nWe are going to use SVC as our baseline model, and we'll try to improve accuracy in future models.","metadata":{}},{"cell_type":"markdown","source":"Finding the best parameters with grid search","metadata":{}},{"cell_type":"code","source":"def find_best_hyperparameters(X_train, y_train):\n\n    # Define the hyperparameter grid\n    param_grid = {\n        'C': [0.1, 1, 10],\n        'kernel': ['linear', 'poly', 'rbf', 'sigmoid'],\n        'gamma': ['scale', 'auto', 0.1, 1, 10],\n    }\n\n    # Create the SVC model\n    svc = SVC(random_state=42)\n\n    # Set up GridSearchCV\n    grid_search = GridSearchCV(\n        estimator=svc,\n        param_grid=param_grid,\n        scoring='accuracy',  # Metric to optimize\n        cv=5,  # 5-fold cross-validation\n        verbose=1,\n        n_jobs=1  # Trying -1 to use all cores throws an error\n    )\n\n    # Fit GridSearchCV to the data\n    grid_search.fit(X_train, y_train)\n\n    # Return the best parameters and the corresponding score\n    return grid_search.best_params_, grid_search.best_score_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:24.408170Z","iopub.execute_input":"2024-12-02T02:00:24.408589Z","iopub.status.idle":"2024-12-02T02:00:24.415728Z","shell.execute_reply.started":"2024-12-02T02:00:24.408552Z","shell.execute_reply":"2024-12-02T02:00:24.414507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_params, best_score = find_best_hyperparameters(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:37:44.997242Z","iopub.execute_input":"2024-12-01T22:37:44.997614Z","iopub.status.idle":"2024-12-01T22:49:53.108541Z","shell.execute_reply.started":"2024-12-01T22:37:44.997581Z","shell.execute_reply":"2024-12-01T22:49:53.107278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"best_params, best_score","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:12:58.355604Z","iopub.execute_input":"2024-12-02T01:12:58.356191Z","iopub.status.idle":"2024-12-02T01:12:58.384923Z","shell.execute_reply.started":"2024-12-02T01:12:58.356127Z","shell.execute_reply":"2024-12-02T01:12:58.383158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"svc_model = SVC(C=1, gamma='auto', kernel='rbf', random_state=42)\n\nsvc_model.fit(X_train, y_train)\n\nsvc_model.score(X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:27.429789Z","iopub.execute_input":"2024-12-02T02:00:27.430213Z","iopub.status.idle":"2024-12-02T02:00:28.213093Z","shell.execute_reply.started":"2024-12-02T02:00:27.430177Z","shell.execute_reply":"2024-12-02T02:00:28.211860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_preds = svc_model.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:13:03.443202Z","iopub.execute_input":"2024-12-02T01:13:03.444216Z","iopub.status.idle":"2024-12-02T01:13:03.601885Z","shell.execute_reply.started":"2024-12-02T01:13:03.444178Z","shell.execute_reply":"2024-12-02T01:13:03.600860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conf_matrix = confusion_matrix(y_preds, y_test)\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix,\n                              display_labels=svc_model.classes_)\nsns.set_style(\"whitegrid\", {'axes.grid' : False})\ndisp.plot()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:13:03.698575Z","iopub.execute_input":"2024-12-02T01:13:03.699442Z","iopub.status.idle":"2024-12-02T01:13:04.010278Z","shell.execute_reply.started":"2024-12-02T01:13:03.699400Z","shell.execute_reply":"2024-12-02T01:13:04.008875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"get_pca_results(svc_model, X_train, y_train, X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:22:54.749838Z","iopub.execute_input":"2024-12-02T01:22:54.750841Z","iopub.status.idle":"2024-12-02T01:22:58.264870Z","shell.execute_reply.started":"2024-12-02T01:22:54.750797Z","shell.execute_reply":"2024-12-02T01:22:58.263442Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# LogisticRegression model","metadata":{}},{"cell_type":"code","source":"lor = LogisticRegressionCV(penalty='l2',solver='lbfgs', random_state=42, max_iter=10000)\n\nlor.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:49:58.596634Z","iopub.execute_input":"2024-12-01T22:49:58.597047Z","iopub.status.idle":"2024-12-01T22:50:25.162639Z","shell.execute_reply.started":"2024-12-01T22:49:58.596999Z","shell.execute_reply":"2024-12-01T22:50:25.160752Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score = lor.score(X_test, y_test)\nscore","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:50:25.164024Z","iopub.execute_input":"2024-12-01T22:50:25.165357Z","iopub.status.idle":"2024-12-01T22:50:25.189574Z","shell.execute_reply.started":"2024-12-01T22:50:25.165269Z","shell.execute_reply":"2024-12-01T22:50:25.188125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_preds = lor.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:50:25.196019Z","iopub.execute_input":"2024-12-01T22:50:25.197377Z","iopub.status.idle":"2024-12-01T22:50:25.222768Z","shell.execute_reply.started":"2024-12-01T22:50:25.197290Z","shell.execute_reply":"2024-12-01T22:50:25.220962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conf_matrix = confusion_matrix(y_preds, y_test)\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix,\n                              display_labels=lor.classes_)\n\ndisp.plot()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:50:25.224376Z","iopub.execute_input":"2024-12-01T22:50:25.224945Z","iopub.status.idle":"2024-12-01T22:50:25.606908Z","shell.execute_reply.started":"2024-12-01T22:50:25.224883Z","shell.execute_reply":"2024-12-01T22:50:25.605559Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The logistic regression model performed a little better than our baseline. Let's check if we can improve this performance by applying PCA to the dataset.","metadata":{}},{"cell_type":"code","source":"get_pca_results(lor, X_train, y_train, X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:50:25.608407Z","iopub.execute_input":"2024-12-01T22:50:25.608784Z","iopub.status.idle":"2024-12-01T22:51:01.869123Z","shell.execute_reply.started":"2024-12-01T22:50:25.608751Z","shell.execute_reply":"2024-12-01T22:51:01.868122Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Gradient Boosting Classifier","metadata":{}},{"cell_type":"code","source":"gbc = GradientBoostingClassifier(loss='log_loss', learning_rate=0.1, n_estimators=100, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:51:01.870809Z","iopub.execute_input":"2024-12-01T22:51:01.871231Z","iopub.status.idle":"2024-12-01T22:51:01.877097Z","shell.execute_reply.started":"2024-12-01T22:51:01.871186Z","shell.execute_reply":"2024-12-01T22:51:01.875768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gbc.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:51:01.878406Z","iopub.execute_input":"2024-12-01T22:51:01.878897Z","iopub.status.idle":"2024-12-01T22:51:09.275479Z","shell.execute_reply.started":"2024-12-01T22:51:01.878842Z","shell.execute_reply":"2024-12-01T22:51:09.274214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"gbc.score(X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:51:09.277266Z","iopub.execute_input":"2024-12-01T22:51:09.277783Z","iopub.status.idle":"2024-12-01T22:51:09.300661Z","shell.execute_reply.started":"2024-12-01T22:51:09.277730Z","shell.execute_reply":"2024-12-01T22:51:09.299115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_preds = gbc.predict(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:51:09.302160Z","iopub.execute_input":"2024-12-01T22:51:09.302521Z","iopub.status.idle":"2024-12-01T22:51:09.320897Z","shell.execute_reply.started":"2024-12-01T22:51:09.302478Z","shell.execute_reply":"2024-12-01T22:51:09.319478Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"conf_matrix = confusion_matrix(y_preds, y_test)\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix,\n                              display_labels=gbc.classes_)\n\ndisp.plot()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:51:09.322339Z","iopub.execute_input":"2024-12-01T22:51:09.322743Z","iopub.status.idle":"2024-12-01T22:51:09.619024Z","shell.execute_reply.started":"2024-12-01T22:51:09.322675Z","shell.execute_reply":"2024-12-01T22:51:09.617577Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The gradient boosting classifier shows a lower performance in our validation data in comparison to our baseline model.","metadata":{}},{"cell_type":"code","source":"get_pca_results(gbc, X_train, y_train, X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:51:09.620897Z","iopub.execute_input":"2024-12-01T22:51:09.621407Z","iopub.status.idle":"2024-12-01T22:51:59.505885Z","shell.execute_reply.started":"2024-12-01T22:51:09.621358Z","shell.execute_reply":"2024-12-01T22:51:59.504557Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Random Forest Classifier","metadata":{}},{"cell_type":"code","source":"param_grid = { \n    'n_estimators': [25, 50, 100, 150], \n    'max_features': ['sqrt', 'log2', None], \n    'max_depth': [3, 6, 9], \n    'max_leaf_nodes': [3, 6, 9], \n} \n\ngrid_search = GridSearchCV(RandomForestClassifier(), \n                           param_grid=param_grid) \ngrid_search.fit(X_train, y_train) \nprint(grid_search.best_estimator_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:51:59.507758Z","iopub.execute_input":"2024-12-01T22:51:59.508106Z","iopub.status.idle":"2024-12-01T22:56:32.708243Z","shell.execute_reply.started":"2024-12-01T22:51:59.508075Z","shell.execute_reply":"2024-12-01T22:56:32.706861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rfc = grid_search.best_estimator_\n\nrfc.fit(X_train, y_train)\nrfc.score(X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:56:32.709759Z","iopub.execute_input":"2024-12-01T22:56:32.710147Z","iopub.status.idle":"2024-12-01T22:56:33.183017Z","shell.execute_reply.started":"2024-12-01T22:56:32.710112Z","shell.execute_reply":"2024-12-01T22:56:33.181770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_preds = rfc.predict(X_test)\nconf_matrix = confusion_matrix(y_preds, y_test)\ndisp = ConfusionMatrixDisplay(confusion_matrix=conf_matrix,\n                              display_labels=rfc.classes_)\n\ndisp.plot()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:56:33.184793Z","iopub.execute_input":"2024-12-01T22:56:33.185273Z","iopub.status.idle":"2024-12-01T22:56:33.566428Z","shell.execute_reply.started":"2024-12-01T22:56:33.185224Z","shell.execute_reply":"2024-12-01T22:56:33.565132Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The random forest classifier is not able to learn on the characteristics of the tabular data on this problem.","metadata":{}},{"cell_type":"code","source":"get_pca_results(rfc, X_train, y_train, X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:56:33.568019Z","iopub.execute_input":"2024-12-01T22:56:33.568392Z","iopub.status.idle":"2024-12-01T22:56:37.219311Z","shell.execute_reply.started":"2024-12-01T22:56:33.568359Z","shell.execute_reply":"2024-12-01T22:56:37.218247Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Neural networks\n\nSo far, the best performing model has been the LogisticClassifier, is there a chance that we can get a better performance using neural networks?","metadata":{}},{"cell_type":"code","source":"import keras\nimport keras_tuner\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Input\nfrom tensorflow.keras.utils import to_categorical","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:56:37.220564Z","iopub.execute_input":"2024-12-01T22:56:37.220849Z","iopub.status.idle":"2024-12-01T22:56:51.713073Z","shell.execute_reply.started":"2024-12-01T22:56:37.220821Z","shell.execute_reply":"2024-12-01T22:56:51.711564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Input((X_train.shape[1],)))\nmodel.add(Dense(64, activation='relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(32, activation='relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(4, activation='softmax'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:56:51.715000Z","iopub.execute_input":"2024-12-01T22:56:51.715830Z","iopub.status.idle":"2024-12-01T22:56:51.828905Z","shell.execute_reply.started":"2024-12-01T22:56:51.715792Z","shell.execute_reply":"2024-12-01T22:56:51.827745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer = keras.optimizers.Adam(learning_rate=0.001)\nmodel.compile(optimizer=optimizer, loss='sparse_categorical_crossentropy', metrics=['accuracy'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:56:51.830621Z","iopub.execute_input":"2024-12-01T22:56:51.831076Z","iopub.status.idle":"2024-12-01T22:56:51.849247Z","shell.execute_reply.started":"2024-12-01T22:56:51.831039Z","shell.execute_reply":"2024-12-01T22:56:51.847801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.fit(X_train, y_train, validation_split=0.2, epochs=100, verbose=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:56:51.850697Z","iopub.execute_input":"2024-12-01T22:56:51.851032Z","iopub.status.idle":"2024-12-01T22:57:15.077790Z","shell.execute_reply.started":"2024-12-01T22:56:51.851000Z","shell.execute_reply":"2024-12-01T22:57:15.076515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:57:15.079079Z","iopub.execute_input":"2024-12-01T22:57:15.079408Z","iopub.status.idle":"2024-12-01T22:57:15.205351Z","shell.execute_reply.started":"2024-12-01T22:57:15.079376Z","shell.execute_reply":"2024-12-01T22:57:15.204187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_predictions = model.predict(X_test)\npred_test = np.zeros(X_test.shape[0])\n\nfor id in range(X_test.shape[0]):\n    pred_test[id] = np.argmax( test_predictions[id] )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:57:15.206780Z","iopub.execute_input":"2024-12-01T22:57:15.207064Z","iopub.status.idle":"2024-12-01T22:57:15.403862Z","shell.execute_reply.started":"2024-12-01T22:57:15.207036Z","shell.execute_reply":"2024-12-01T22:57:15.402583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"con = confusion_matrix( y_test , pred_test )\ndisp = ConfusionMatrixDisplay( confusion_matrix = con,  display_labels = [\"0\",\"1\",\"2\",\"3\"] ).plot()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-01T22:57:15.405471Z","iopub.execute_input":"2024-12-01T22:57:15.405894Z","iopub.status.idle":"2024-12-01T22:57:15.958177Z","shell.execute_reply.started":"2024-12-01T22:57:15.405847Z","shell.execute_reply":"2024-12-01T22:57:15.956816Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As we see, a neural network does not perform better than a classic learning machine algo for this problem.","metadata":{}},{"cell_type":"markdown","source":"# Model selection\n\nWe have decided that the best model for this problem is the SVC, as it is the one that obtained the highest validation accuracy. And we are going to keep the same attributes with no use of PCA, as the accuracy is higher and the difference in training time is not big enough to justify.\n\nNow we're going to apply KFold cross validation to ensure the model is capable of learning on the data.","metadata":{}},{"cell_type":"code","source":"scores = cross_val_score(svc_model, X, y, scoring='accuracy', cv=KFold(n_splits=10))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:29:23.749261Z","iopub.execute_input":"2024-12-02T01:29:23.749988Z","iopub.status.idle":"2024-12-02T01:29:32.069554Z","shell.execute_reply.started":"2024-12-02T01:29:23.749931Z","shell.execute_reply":"2024-12-02T01:29:32.068425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Accuracy: %.3f ,\\nStandard Deviations :%.3f' %\n      (np.mean(scores), np.std(scores)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T01:31:03.442809Z","iopub.execute_input":"2024-12-02T01:31:03.443948Z","iopub.status.idle":"2024-12-02T01:31:03.450326Z","shell.execute_reply.started":"2024-12-02T01:31:03.443883Z","shell.execute_reply":"2024-12-02T01:31:03.449074Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The model shows an accuracy similar to the one found during model selection, and the standar deviation is close to 0, which tells us the model is equally good across the data.","metadata":{}},{"cell_type":"markdown","source":"# Submission\n\nNow that we have cleaned the data, applied preprocessing to it, tried different models with different hyperparameters, different attributes and finally selected the best one based on performance, we can create a submission file for the competition.","metadata":{}},{"cell_type":"code","source":"# Final training\nsvc_model.fit(X_train, y_train)\nsvc_model.score(X_test, y_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:37.882953Z","iopub.execute_input":"2024-12-02T02:00:37.883542Z","iopub.status.idle":"2024-12-02T02:00:38.662351Z","shell.execute_reply.started":"2024-12-02T02:00:37.883488Z","shell.execute_reply":"2024-12-02T02:00:38.661210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df = pd.DataFrame({\"id\": test[\"id\"]})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:38.664306Z","iopub.execute_input":"2024-12-02T02:00:38.664654Z","iopub.status.idle":"2024-12-02T02:00:38.670610Z","shell.execute_reply.started":"2024-12-02T02:00:38.664618Z","shell.execute_reply":"2024-12-02T02:00:38.669489Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = svc_model.predict(test.drop(\"id\", axis=1))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:38.672070Z","iopub.execute_input":"2024-12-02T02:00:38.672514Z","iopub.status.idle":"2024-12-02T02:00:38.693292Z","shell.execute_reply.started":"2024-12-02T02:00:38.672456Z","shell.execute_reply":"2024-12-02T02:00:38.692135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df['sii'] = predictions.astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-02T02:00:38.993800Z","iopub.execute_input":"2024-12-02T02:00:38.994218Z","iopub.status.idle":"2024-12-02T02:00:38.999878Z","shell.execute_reply.started":"2024-12-02T02:00:38.994183Z","shell.execute_reply":"2024-12-02T02:00:38.998754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-12-02T02:02:30.449458Z","iopub.execute_input":"2024-12-02T02:02:30.449965Z","iopub.status.idle":"2024-12-02T02:02:30.457816Z","shell.execute_reply.started":"2024-12-02T02:02:30.449926Z","shell.execute_reply":"2024-12-02T02:02:30.456528Z"},"trusted":true},"outputs":[],"execution_count":null}]}