{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-27T00:21:01.207630Z","iopub.execute_input":"2024-10-27T00:21:01.208072Z","iopub.status.idle":"2024-10-27T00:21:04.341145Z","shell.execute_reply.started":"2024-10-27T00:21:01.208026Z","shell.execute_reply":"2024-10-27T00:21:04.339724Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib as mpl\nfrom matplotlib import pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:26:35.671882Z","iopub.execute_input":"2024-10-27T00:26:35.672365Z","iopub.status.idle":"2024-10-27T00:26:35.677732Z","shell.execute_reply.started":"2024-10-27T00:26:35.672315Z","shell.execute_reply":"2024-10-27T00:26:35.676541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:25:18.437505Z","iopub.execute_input":"2024-10-27T00:25:18.437919Z","iopub.status.idle":"2024-10-27T00:25:18.491719Z","shell.execute_reply.started":"2024-10-27T00:25:18.437880Z","shell.execute_reply":"2024-10-27T00:25:18.490465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Boxplot vertical\nfigure,axes = plt.subplots(figsize=(10, 5))\n\nsns.boxplot(x=df['sii'], y=df['Basic_Demos-Age'], data=df , hue='Basic_Demos-Sex')\n\naxes.set_title('Sexo',fontsize = 10)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:26:53.644004Z","iopub.execute_input":"2024-10-27T00:26:53.644998Z","iopub.status.idle":"2024-10-27T00:26:54.406659Z","shell.execute_reply.started":"2024-10-27T00:26:53.644948Z","shell.execute_reply":"2024-10-27T00:26:54.405534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info(verbose=True)\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:27:23.747417Z","iopub.execute_input":"2024-10-27T00:27:23.747861Z","iopub.status.idle":"2024-10-27T00:27:23.780904Z","shell.execute_reply.started":"2024-10-27T00:27:23.747817Z","shell.execute_reply":"2024-10-27T00:27:23.779642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"campos=['sii','id','Basic_Demos-Age','Basic_Demos-Sex','Physical-BMI','Physical-Height','Physical-Weight','PreInt_EduHx-computerinternet_hoursday','PCIAT-PCIAT_Total']\ndf=df[campos]\ndf.info(verbose=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:27:52.776415Z","iopub.execute_input":"2024-10-27T00:27:52.777298Z","iopub.status.idle":"2024-10-27T00:27:52.793892Z","shell.execute_reply.started":"2024-10-27T00:27:52.777220Z","shell.execute_reply":"2024-10-27T00:27:52.792640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp= df\ncorr_matrix = temp.select_dtypes(include=['float64', 'int']).corr(method='pearson')\ncorr_matrix\nfig, ax = plt.subplots(nrows=1, ncols=1, figsize=(10, 10))\n\nsns.heatmap(\n    corr_matrix,\n    annot     = True,\n    cbar      = False,\n    annot_kws = {\"size\": 8},\n    vmin      = -1,\n    vmax      = 1,\n    center    = 0,\n    cmap      = sns.diverging_palette(20, 220, n=200),\n    square    = True,\n    ax        = ax\n)\n\nax.set_xticklabels(\n    ax.get_xticklabels(),\n    rotation = 45,\n    horizontalalignment = 'right',\n)\n\nax.tick_params(labelsize = 10)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:28:34.668597Z","iopub.execute_input":"2024-10-27T00:28:34.669612Z","iopub.status.idle":"2024-10-27T00:28:35.194040Z","shell.execute_reply.started":"2024-10-27T00:28:34.669545Z","shell.execute_reply":"2024-10-27T00:28:35.192860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Contar el numero de datos nulos\ndf[df.isna().any(axis='columns')].count()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:28:54.591935Z","iopub.execute_input":"2024-10-27T00:28:54.592437Z","iopub.status.idle":"2024-10-27T00:28:54.606464Z","shell.execute_reply.started":"2024-10-27T00:28:54.592383Z","shell.execute_reply":"2024-10-27T00:28:54.605089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# los datos faltantes estan representados por '?'\ndf = df.replace('?',np.nan)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:29:22.583312Z","iopub.execute_input":"2024-10-27T00:29:22.583734Z","iopub.status.idle":"2024-10-27T00:29:22.590972Z","shell.execute_reply.started":"2024-10-27T00:29:22.583694Z","shell.execute_reply":"2024-10-27T00:29:22.589728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df.isna().any(axis='columns')].count()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:29:41.308502Z","iopub.execute_input":"2024-10-27T00:29:41.309568Z","iopub.status.idle":"2024-10-27T00:29:41.324800Z","shell.execute_reply.started":"2024-10-27T00:29:41.309504Z","shell.execute_reply":"2024-10-27T00:29:41.323418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.dropna(subset=['sii'])\ndf = df.dropna(subset=['Basic_Demos-Age'])\ndf = df.dropna(subset=['Physical-BMI'])\ndf = df.dropna(subset=['PreInt_EduHx-computerinternet_hoursday'])\n\n\n#df = df.dropna(subset=['Physical-Waist_Circumference'])\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:31:36.755099Z","iopub.execute_input":"2024-10-27T00:31:36.755595Z","iopub.status.idle":"2024-10-27T00:31:36.776788Z","shell.execute_reply.started":"2024-10-27T00:31:36.755548Z","shell.execute_reply":"2024-10-27T00:31:36.775644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ver todas edades registradas sin repetición\ndf['Basic_Demos-Age'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:31:58.349660Z","iopub.execute_input":"2024-10-27T00:31:58.350158Z","iopub.status.idle":"2024-10-27T00:31:58.358374Z","shell.execute_reply.started":"2024-10-27T00:31:58.350114Z","shell.execute_reply":"2024-10-27T00:31:58.356947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_categoricas = ['sii','Basic_Demos-Sex','PreInt_EduHx-computerinternet_hoursday','PCIAT-PCIAT_Total']\n\ndf[cols_categoricas] = df[cols_categoricas].astype(\"category\")","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:32:18.358120Z","iopub.execute_input":"2024-10-27T00:32:18.358590Z","iopub.status.idle":"2024-10-27T00:32:18.370529Z","shell.execute_reply.started":"2024-10-27T00:32:18.358545Z","shell.execute_reply":"2024-10-27T00:32:18.369364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_numericas = ['Basic_Demos-Age','Physical-BMI','Physical-Height','Physical-Weight']\n\ndf[cols_numericas] = df[cols_numericas].astype(\"float\")","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:32:49.985636Z","iopub.execute_input":"2024-10-27T00:32:49.986112Z","iopub.status.idle":"2024-10-27T00:32:49.994629Z","shell.execute_reply.started":"2024-10-27T00:32:49.986067Z","shell.execute_reply":"2024-10-27T00:32:49.993456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:47:09.494921Z","iopub.execute_input":"2024-10-27T00:47:09.495388Z","iopub.status.idle":"2024-10-27T00:47:09.520070Z","shell.execute_reply.started":"2024-10-27T00:47:09.495341Z","shell.execute_reply":"2024-10-27T00:47:09.518742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=df.drop('id', axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:47:32.455237Z","iopub.execute_input":"2024-10-27T00:47:32.455709Z","iopub.status.idle":"2024-10-27T00:47:32.463164Z","shell.execute_reply.started":"2024-10-27T00:47:32.455664Z","shell.execute_reply":"2024-10-27T00:47:32.461875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:47:59.505139Z","iopub.execute_input":"2024-10-27T00:47:59.505627Z","iopub.status.idle":"2024-10-27T00:47:59.534419Z","shell.execute_reply.started":"2024-10-27T00:47:59.505577Z","shell.execute_reply":"2024-10-27T00:47:59.533190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_parquet('proyecto_processed.parquet',\n                      index=False)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:48:19.977128Z","iopub.execute_input":"2024-10-27T00:48:19.977676Z","iopub.status.idle":"2024-10-27T00:48:20.075228Z","shell.execute_reply.started":"2024-10-27T00:48:19.977627Z","shell.execute_reply":"2024-10-27T00:48:20.074053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visulizar la distribucion de los datos por categoria de la variable sii\n\ndf[\"sii\"].value_counts().plot(kind=\"bar\",\n                                           color=['skyblue', 'orange'])","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:49:00.104124Z","iopub.execute_input":"2024-10-27T00:49:00.104944Z","iopub.status.idle":"2024-10-27T00:49:00.408045Z","shell.execute_reply.started":"2024-10-27T00:49:00.104896Z","shell.execute_reply":"2024-10-27T00:49:00.406874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp= df\ncorr_matrix = temp.select_dtypes(include=['float64', 'category']).corr(method='pearson')\ncorr_matrix\nfig, ax = plt.subplots(nrows=1, ncols=1, figsize=(8, 8))\n\nsns.heatmap(\n    corr_matrix,\n    annot     = True,\n    cbar      = False,\n    annot_kws = {\"size\": 8},\n    vmin      = -1,\n    vmax      = 1,\n    center    = 0,\n    cmap      = sns.diverging_palette(20, 220, n=200),\n    square    = True,\n    ax        = ax\n)\n\nax.set_xticklabels(\n    ax.get_xticklabels(),\n    rotation = 45,\n    horizontalalignment = 'right',\n)\n\nax.tick_params(labelsize = 10)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:49:43.063362Z","iopub.execute_input":"2024-10-27T00:49:43.063795Z","iopub.status.idle":"2024-10-27T00:49:43.605461Z","shell.execute_reply.started":"2024-10-27T00:49:43.063754Z","shell.execute_reply":"2024-10-27T00:49:43.604170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_numericas = ['Basic_Demos-Age','Physical-BMI','Physical-Height','Physical-Weight']\ncols_categoricas_ord = ['sii','Basic_Demos-Sex','PreInt_EduHx-computerinternet_hoursday','PCIAT-PCIAT_Total']","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:50:16.487012Z","iopub.execute_input":"2024-10-27T00:50:16.487494Z","iopub.status.idle":"2024-10-27T00:50:16.493062Z","shell.execute_reply.started":"2024-10-27T00:50:16.487448Z","shell.execute_reply":"2024-10-27T00:50:16.491788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.compose import ColumnTransformer","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:50:34.159784Z","iopub.execute_input":"2024-10-27T00:50:34.160245Z","iopub.status.idle":"2024-10-27T00:50:34.801445Z","shell.execute_reply.started":"2024-10-27T00:50:34.160200Z","shell.execute_reply":"2024-10-27T00:50:34.800178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"numeric_pipe = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    ])\n\ncategorical_ord_pipe = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OrdinalEncoder())])\n\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('numericas', numeric_pipe, cols_numericas),\n        ('categoricas ordinales', categorical_ord_pipe, cols_categoricas_ord)\n])\nnumeric_pipe = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='median')),\n    ])\n\ncategorical_ord_pipe = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OrdinalEncoder())])\n\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('numericas', numeric_pipe, cols_numericas),\n        ('categoricas ordinales', categorical_ord_pipe, cols_categoricas_ord)\n])\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:53:53.659851Z","iopub.execute_input":"2024-10-27T00:53:53.660280Z","iopub.status.idle":"2024-10-27T00:53:53.666913Z","shell.execute_reply.started":"2024-10-27T00:53:53.660220Z","shell.execute_reply":"2024-10-27T00:53:53.665681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessor ","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:55:00.755930Z","iopub.execute_input":"2024-10-27T00:55:00.756361Z","iopub.status.idle":"2024-10-27T00:55:00.788093Z","shell.execute_reply.started":"2024-10-27T00:55:00.756319Z","shell.execute_reply":"2024-10-27T00:55:00.787055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_features = df.drop('sii', axis='columns')\nY_target = df['sii']\n\nx_train, x_test, y_train, y_test = train_test_split(X_features,\n                                                    Y_target,\n                                                    test_size=0.2,\n                                                    stratify=Y_target, random_state=0)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:55:33.095311Z","iopub.execute_input":"2024-10-27T00:55:33.095735Z","iopub.status.idle":"2024-10-27T00:55:33.112950Z","shell.execute_reply.started":"2024-10-27T00:55:33.095692Z","shell.execute_reply":"2024-10-27T00:55:33.111815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# conjunto de datos de entrenamiento\nx_train.shape, y_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:55:59.218948Z","iopub.execute_input":"2024-10-27T00:55:59.219667Z","iopub.status.idle":"2024-10-27T00:55:59.226526Z","shell.execute_reply.started":"2024-10-27T00:55:59.219622Z","shell.execute_reply":"2024-10-27T00:55:59.225501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nmodel = LogisticRegression()\nmodel.fit(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:56:34.627133Z","iopub.execute_input":"2024-10-27T00:56:34.628120Z","iopub.status.idle":"2024-10-27T00:56:34.712686Z","shell.execute_reply.started":"2024-10-27T00:56:34.628072Z","shell.execute_reply":"2024-10-27T00:56:34.711588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, classification_report\n\ny_pred = model.predict(x_test)\n\nprint(f\"Precisión: {accuracy_score(y_test, y_pred)}\")\nprint(classification_report(y_test, y_pred))\n","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:56:54.427044Z","iopub.execute_input":"2024-10-27T00:56:54.427506Z","iopub.status.idle":"2024-10-27T00:56:54.449517Z","shell.execute_reply.started":"2024-10-27T00:56:54.427449Z","shell.execute_reply":"2024-10-27T00:56:54.448200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"comparacion_df = pd.DataFrame({\n                'Valor Real': y_test.values,\n                'Predicción': y_pred\n                }, index = y_test.index)\ncomparacion_df","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:57:13.041518Z","iopub.execute_input":"2024-10-27T00:57:13.042105Z","iopub.status.idle":"2024-10-27T00:57:13.062181Z","shell.execute_reply.started":"2024-10-27T00:57:13.042044Z","shell.execute_reply":"2024-10-27T00:57:13.060772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Graficas de evaluación para modelos de clasificación\nfrom sklearn.metrics import PrecisionRecallDisplay\nfrom sklearn.metrics import ConfusionMatrixDisplay\nfrom sklearn.metrics import roc_curve\nfrom sklearn.metrics import RocCurveDisplay","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:57:32.938969Z","iopub.execute_input":"2024-10-27T00:57:32.939393Z","iopub.status.idle":"2024-10-27T00:57:32.944840Z","shell.execute_reply.started":"2024-10-27T00:57:32.939354Z","shell.execute_reply":"2024-10-27T00:57:32.943631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ConfusionMatrixDisplay.from_predictions(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2024-10-27T00:57:54.772773Z","iopub.execute_input":"2024-10-27T00:57:54.773877Z","iopub.status.idle":"2024-10-27T00:57:55.146236Z","shell.execute_reply.started":"2024-10-27T00:57:54.773815Z","shell.execute_reply":"2024-10-27T00:57:55.144930Z"},"trusted":true},"execution_count":null,"outputs":[]}]}