{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Import Libraries**","metadata":{}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport torch\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom sklearn.model_selection import GridSearchCV, StratifiedKFold\nfrom sklearn.metrics import classification_report, confusion_matrix, roc_auc_score, RocCurveDisplay, accuracy_score\nfrom sklearn.metrics import roc_curve, auc\nfrom sklearn.preprocessing import StandardScaler, label_binarize, PolynomialFeatures\nfrom sklearn import preprocessing\nfrom sklearn.decomposition import PCA\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error, mean_absolute_error, r2_score\nfrom xgboost import XGBClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.ensemble import VotingClassifier, RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.compose import ColumnTransformer","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-05T08:42:46.487978Z","iopub.execute_input":"2024-12-05T08:42:46.488389Z","iopub.status.idle":"2024-12-05T08:42:46.500782Z","shell.execute_reply.started":"2024-12-05T08:42:46.488354Z","shell.execute_reply":"2024-12-05T08:42:46.499801Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Read Dataset**","metadata":{}},{"cell_type":"code","source":"train_ds = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv', index_col='id')\n\ntest_ds = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv', index_col='id')\n\ndata_dictionary = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:08:38.166234Z","iopub.execute_input":"2024-12-05T07:08:38.166618Z","iopub.status.idle":"2024-12-05T07:08:38.262810Z","shell.execute_reply.started":"2024-12-05T07:08:38.166586Z","shell.execute_reply":"2024-12-05T07:08:38.261926Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Analyze Dataset**","metadata":{}},{"cell_type":"code","source":"train_ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:08:50.338581Z","iopub.execute_input":"2024-12-05T07:08:50.339003Z","iopub.status.idle":"2024-12-05T07:08:50.383205Z","shell.execute_reply.started":"2024-12-05T07:08:50.338968Z","shell.execute_reply":"2024-12-05T07:08:50.382204Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:24:20.139066Z","iopub.execute_input":"2024-12-05T07:24:20.139458Z","iopub.status.idle":"2024-12-05T07:24:20.175285Z","shell.execute_reply.started":"2024-12-05T07:24:20.139422Z","shell.execute_reply":"2024-12-05T07:24:20.174208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ds.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:29:04.927767Z","iopub.execute_input":"2024-12-05T07:29:04.928586Z","iopub.status.idle":"2024-12-05T07:29:04.966945Z","shell.execute_reply.started":"2024-12-05T07:29:04.928547Z","shell.execute_reply":"2024-12-05T07:29:04.965608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_ds.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:29:28.135939Z","iopub.execute_input":"2024-12-05T07:29:28.137099Z","iopub.status.idle":"2024-12-05T07:29:28.154961Z","shell.execute_reply.started":"2024-12-05T07:29:28.137041Z","shell.execute_reply":"2024-12-05T07:29:28.153630Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Change the values on columns with Dtype object to numbers. In this case, they are all columns with the names of seasons.","metadata":{}},{"cell_type":"code","source":"train_cat_columns = train_ds.select_dtypes(exclude = 'number').columns\n\nfor season in train_cat_columns:\n    train_ds[season] = train_ds[season].replace({'Spring':1, 'Summer':2, 'Fall':3, 'Winter':4})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:44:25.390326Z","iopub.execute_input":"2024-12-05T07:44:25.391341Z","iopub.status.idle":"2024-12-05T07:44:25.423859Z","shell.execute_reply.started":"2024-12-05T07:44:25.391266Z","shell.execute_reply":"2024-12-05T07:44:25.422807Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_cat_columns = test_ds.select_dtypes(exclude = 'number').columns\n\nfor season in test_cat_columns:\n    test_ds[season] = test_ds[season].replace({'Spring':1, 'Summer':2, 'Fall':3, 'Winter':4})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:45:20.036838Z","iopub.execute_input":"2024-12-05T07:45:20.037246Z","iopub.status.idle":"2024-12-05T07:45:20.053494Z","shell.execute_reply.started":"2024-12-05T07:45:20.037211Z","shell.execute_reply":"2024-12-05T07:45:20.052219Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Selecting the data**","metadata":{}},{"cell_type":"markdown","source":"Eliminating all the rows that have a null value on the target column 'sii'. Eliminating the columns that aren't on test.csv. Eliminating all the columns that have more than 50% of the remaining rows with null values, and finally, we use the .corr function of pandas, which uses Pearson correlation, to see which of the remaining columns have the most effect on 'PCIAT-PCIAT_Total' from which 'sii' originates from.","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"## Eliminating null rows","metadata":{}},{"cell_type":"code","source":"null = train_ds.isna().sum().sort_values(ascending = False).head(70)\nnull = pd.DataFrame(null)\nnull = null.rename(columns= {0:'Missing'})\nnull.style.background_gradient(cmap='YlOrRd')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:45:31.230881Z","iopub.execute_input":"2024-12-05T07:45:31.231285Z","iopub.status.idle":"2024-12-05T07:45:31.256298Z","shell.execute_reply.started":"2024-12-05T07:45:31.231248Z","shell.execute_reply":"2024-12-05T07:45:31.255180Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ds_usable = train_ds[train_ds['sii'].notnull()]\n\nnull = train_ds_usable.isna().sum().sort_values(ascending = False).head(70)\nnull = pd.DataFrame(null)\nnull = null.rename(columns= {0:'Missing'})\nnull.style.background_gradient(cmap='YlOrRd')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:47:01.247349Z","iopub.execute_input":"2024-12-05T07:47:01.247812Z","iopub.status.idle":"2024-12-05T07:47:01.270278Z","shell.execute_reply.started":"2024-12-05T07:47:01.247774Z","shell.execute_reply":"2024-12-05T07:47:01.269161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ds_usable.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:45:46.105761Z","iopub.execute_input":"2024-12-05T07:45:46.106171Z","iopub.status.idle":"2024-12-05T07:45:46.112723Z","shell.execute_reply.started":"2024-12-05T07:45:46.106134Z","shell.execute_reply":"2024-12-05T07:45:46.111695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ds_usable.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:47:33.438648Z","iopub.execute_input":"2024-12-05T07:47:33.439792Z","iopub.status.idle":"2024-12-05T07:47:33.456552Z","shell.execute_reply.started":"2024-12-05T07:47:33.439727Z","shell.execute_reply":"2024-12-05T07:47:33.455268Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Eliminate rows that are not in test.csv","metadata":{}},{"cell_type":"code","source":"PCIAT_cols = [val for val in train_ds_usable.columns[train_ds_usable.columns.str.contains('PCIAT')]]\nprint('Number of PCIAT features = ' , len(PCIAT_cols))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:50:25.470350Z","iopub.execute_input":"2024-12-05T07:50:25.470780Z","iopub.status.idle":"2024-12-05T07:50:25.477544Z","shell.execute_reply.started":"2024-12-05T07:50:25.470735Z","shell.execute_reply":"2024-12-05T07:50:25.476274Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PCIAT_cols.remove('PCIAT-PCIAT_Total')\ntrain_ds_usable = train_ds_usable.drop(columns = PCIAT_cols)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:51:08.557338Z","iopub.execute_input":"2024-12-05T07:51:08.557790Z","iopub.status.idle":"2024-12-05T07:51:08.565555Z","shell.execute_reply.started":"2024-12-05T07:51:08.557746Z","shell.execute_reply":"2024-12-05T07:51:08.564263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ds_usable.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:51:30.082924Z","iopub.execute_input":"2024-12-05T07:51:30.083822Z","iopub.status.idle":"2024-12-05T07:51:30.090336Z","shell.execute_reply.started":"2024-12-05T07:51:30.083768Z","shell.execute_reply":"2024-12-05T07:51:30.089305Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Use Pearson correlation to find which columns to keep","metadata":{}},{"cell_type":"code","source":"corr = pd.DataFrame(train_ds_usable.corr()['PCIAT-PCIAT_Total'].sort_values(ascending = False))\ncorr.style.background_gradient(cmap='YlOrRd')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:52:49.271016Z","iopub.execute_input":"2024-12-05T07:52:49.271568Z","iopub.status.idle":"2024-12-05T07:52:49.321899Z","shell.execute_reply.started":"2024-12-05T07:52:49.271517Z","shell.execute_reply":"2024-12-05T07:52:49.320785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selection = corr[(corr['PCIAT-PCIAT_Total']>.1) | (corr['PCIAT-PCIAT_Total']<-.1)]\nselection = [val for val in selection.index]\nselection.remove('PCIAT-PCIAT_Total')\nselection.remove('sii')\nselection.remove('Physical-BMI')\nselection.remove('SDS-SDS_Total_Raw')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:54:13.531859Z","iopub.execute_input":"2024-12-05T07:54:13.532241Z","iopub.status.idle":"2024-12-05T07:54:13.539921Z","shell.execute_reply.started":"2024-12-05T07:54:13.532211Z","shell.execute_reply":"2024-12-05T07:54:13.538790Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selection","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:54:23.799490Z","iopub.execute_input":"2024-12-05T07:54:23.800081Z","iopub.status.idle":"2024-12-05T07:54:23.808051Z","shell.execute_reply.started":"2024-12-05T07:54:23.800025Z","shell.execute_reply":"2024-12-05T07:54:23.806798Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Eliminate columns with more than 50% of null data","metadata":{}},{"cell_type":"code","source":"half_missing = [val for val in train_ds_usable.columns[train_ds_usable.isnull().sum()>len(train_ds_usable)/2]]\nhalf_missing","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:55:54.853377Z","iopub.execute_input":"2024-12-05T07:55:54.854571Z","iopub.status.idle":"2024-12-05T07:55:54.866241Z","shell.execute_reply.started":"2024-12-05T07:55:54.854512Z","shell.execute_reply":"2024-12-05T07:55:54.865061Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"selection = [i for i in selection if i not in half_missing]\nselection","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T07:56:22.491366Z","iopub.execute_input":"2024-12-05T07:56:22.491907Z","iopub.status.idle":"2024-12-05T07:56:22.500020Z","shell.execute_reply.started":"2024-12-05T07:56:22.491862Z","shell.execute_reply":"2024-12-05T07:56:22.498584Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**And with that we have 15 columns left that will be used for training the model**","metadata":{}},{"cell_type":"code","source":"train_ds_selection = train_ds_usable[selection]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T08:07:35.701965Z","iopub.execute_input":"2024-12-05T08:07:35.703006Z","iopub.status.idle":"2024-12-05T08:07:35.710566Z","shell.execute_reply.started":"2024-12-05T08:07:35.702943Z","shell.execute_reply":"2024-12-05T08:07:35.709166Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ds_selection.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T08:08:02.827681Z","iopub.execute_input":"2024-12-05T08:08:02.828203Z","iopub.status.idle":"2024-12-05T08:08:02.881956Z","shell.execute_reply.started":"2024-12-05T08:08:02.828158Z","shell.execute_reply":"2024-12-05T08:08:02.880659Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Divide data for training and testing the model**","metadata":{}},{"cell_type":"code","source":"X = train_ds_selection\ny = train_ds_usable['sii']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T08:41:13.741759Z","iopub.execute_input":"2024-12-05T08:41:13.742350Z","iopub.status.idle":"2024-12-05T08:41:13.748060Z","shell.execute_reply.started":"2024-12-05T08:41:13.742297Z","shell.execute_reply":"2024-12-05T08:41:13.746900Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T08:42:12.405839Z","iopub.execute_input":"2024-12-05T08:42:12.406495Z","iopub.status.idle":"2024-12-05T08:42:12.417439Z","shell.execute_reply.started":"2024-12-05T08:42:12.406456Z","shell.execute_reply":"2024-12-05T08:42:12.416545Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Preprocessing Pipleline**","metadata":{}},{"cell_type":"markdown","source":"A pipeline to preprocess the data before being used on the model, for the categorical columns remaining the null values are filled by the mode, and for the numerical columns the null values are filled by the mean, and are then used by a standard scaler.","metadata":{}},{"cell_type":"code","source":"categorical_features = ['BIA-BIA_Frame_num', 'FGC-FGC_SRR_Zone', 'FGC-FGC_SRL_Zone']\nnumeric_features = [col for col in X.columns if col not in categorical_features]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T08:43:15.535411Z","iopub.execute_input":"2024-12-05T08:43:15.535862Z","iopub.status.idle":"2024-12-05T08:43:15.541679Z","shell.execute_reply.started":"2024-12-05T08:43:15.535823Z","shell.execute_reply":"2024-12-05T08:43:15.540225Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Classification or Regression?**","metadata":{}},{"cell_type":"markdown","source":"In this case, the problem presented is more represented by a classification problem. Our target column in this case has 4 different values possible that each represent a different class. Although with its great relation to 'PCIAT-PCIAT_Total' it could have also been tackled by making a regression problem with the goal of having a score between 0-100 and then using that number to divide it into the 4 different classes. ","metadata":{}},{"cell_type":"markdown","source":"# **Initial Sample Model**","metadata":{}},{"cell_type":"code","source":"# Pipeline for numerical columns (using mean)\nnumeric_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='mean')),  \n    ('scaler', StandardScaler())  \n])\n\n# Pipeline for categorial columns (using mode)\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent'))  \n])\n\n# Combine both on ColumnTransformer\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numeric_transformer, numeric_features),\n        ('cat', categorical_transformer, categorical_features)\n    ]\n)\n\n# Full pipeline with baseline model\nmodel_pipeline = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('classifier', LogisticRegression(max_iter=1000))\n])\n\n# Train pipeline\nmodel_pipeline.fit(X_train, y_train)\n\n# Predict and evalute\ny_pred = model_pipeline.predict(X_test)\nprint(\"Classification Report:\")\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T08:45:32.885688Z","iopub.execute_input":"2024-12-05T08:45:32.886127Z","iopub.status.idle":"2024-12-05T08:45:32.994983Z","shell.execute_reply.started":"2024-12-05T08:45:32.886087Z","shell.execute_reply":"2024-12-05T08:45:32.994002Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Hyperparameter Tuning**","metadata":{}},{"cell_type":"markdown","source":"Using GridSearchCV to get the best hyperparameters and using those with the RandomForestClassifier model. The RandomForestClassifier is a good model choice for handling non-linear releationships, reduces overfitting and works well when you have many features.","metadata":{}},{"cell_type":"code","source":"# Modelo con Random Forest y búsqueda de hiperparámetros\nparam_grid = {\n    'classifier__n_estimators': [50, 100, 200],\n    'classifier__max_depth': [None, 10, 20, 30],\n    'classifier__min_samples_split': [2, 5, 10]\n}\n\npipeline_rf = Pipeline(steps=[\n    ('preprocessor', preprocessor),\n    ('classifier', RandomForestClassifier(random_state=42))\n])\n\ngrid_search = GridSearchCV(pipeline_rf, param_grid, cv=5, scoring='f1_weighted', n_jobs=-1)\ngrid_search.fit(X_train, y_train)\n\nprint(\"Best Parameters:\", grid_search.best_params_)\nbest_model = grid_search.best_estimator_","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T08:46:36.879670Z","iopub.execute_input":"2024-12-05T08:46:36.880110Z","iopub.status.idle":"2024-12-05T08:47:12.555190Z","shell.execute_reply.started":"2024-12-05T08:46:36.880074Z","shell.execute_reply":"2024-12-05T08:47:12.553931Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Confusion Matrix**","metadata":{}},{"cell_type":"code","source":"y_pred_final = best_model.predict(X_test)\nconf_matrix = confusion_matrix(y_test, y_pred_final)\nsns.heatmap(conf_matrix, annot=True, fmt='d', cmap='Blues')\nplt.title(\"Confusion Matrix\")\nplt.xlabel(\"Predicted\")\nplt.ylabel(\"True\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T08:49:10.539443Z","iopub.execute_input":"2024-12-05T08:49:10.539901Z","iopub.status.idle":"2024-12-05T08:49:10.889146Z","shell.execute_reply.started":"2024-12-05T08:49:10.539866Z","shell.execute_reply":"2024-12-05T08:49:10.888087Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Preparing for submission**","metadata":{}},{"cell_type":"code","source":"test = test_ds[selection]\ntest","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T08:51:25.287855Z","iopub.execute_input":"2024-12-05T08:51:25.288295Z","iopub.status.idle":"2024-12-05T08:51:25.321884Z","shell.execute_reply.started":"2024-12-05T08:51:25.288261Z","shell.execute_reply":"2024-12-05T08:51:25.320684Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_test_pred = best_model.predict(test)\ndf_submit = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv\")\n\nx_sub = df_submit[[\"id\"]].copy()\nx_sub[\"sii\"] = y_test_pred.astype(int)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:01:18.322249Z","iopub.execute_input":"2024-12-05T09:01:18.322731Z","iopub.status.idle":"2024-12-05T09:01:18.352276Z","shell.execute_reply.started":"2024-12-05T09:01:18.322676Z","shell.execute_reply":"2024-12-05T09:01:18.351262Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_sub.to_csv('submission.csv', index=False)\nprint(\"Submission file created: submission.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-05T09:01:39.337584Z","iopub.execute_input":"2024-12-05T09:01:39.337982Z","iopub.status.idle":"2024-12-05T09:01:39.347284Z","shell.execute_reply.started":"2024-12-05T09:01:39.337948Z","shell.execute_reply":"2024-12-05T09:01:39.346320Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Hecho por Ramón Ángel López Castro, Juan Adan Nava Banda y Jazmin Alejandra Escobedo Javalera\nClase Reconocimiento de Patrones (UNISON)\nUsing some of the code and ideas from https://www.kaggle.com/code/sunilkumarmuduli/ml-project-3-problematic-internet-use","metadata":{}}]}