{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Loading The Data Sets","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport os\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom pandas.plotting import andrews_curves\nfrom sklearn.impute import SimpleImputer\n\n\n#Loading the train and test data sets\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ncolumn_names = list(test_data.columns)\ntrain_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n\ntarget = train_data['sii']\ntrain_data = pd.DataFrame(train_data, columns = column_names)\n\ntrain_data['sii'] = target\n\nprint(train_data.columns.difference(test_data.columns))\nprint(train_data.shape)\nprint(test_data.shape)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:33:41.696604Z","iopub.execute_input":"2024-12-13T00:33:41.698405Z","iopub.status.idle":"2024-12-13T00:33:44.995157Z","shell.execute_reply.started":"2024-12-13T00:33:41.698359Z","shell.execute_reply":"2024-12-13T00:33:44.993878Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preprocessing ","metadata":{}},{"cell_type":"code","source":"#Dropping ID columns and but saving test ID column for Kaggle prediction. \nids = test_data['id']\n\ntrain_data = train_data.drop('id', axis=1)\ntest_data = test_data.drop('id', axis=1)\n\n#dropping features with low counts\ncolumns_to_drop = train_data.columns[train_data.count() < 1700]\n\ntrain_data = train_data.drop(columns=columns_to_drop, axis=1, errors='ignore')\ntest_data = test_data.drop(columns=columns_to_drop, axis=1, errors='ignore')\n\n#Using one hot encoding on the categorical data. \n#Cite: https://stackoverflow.com/questions/41973423/typeerror-dataframe-object-is-not-callable\n#Cite: https://www.kaggle.com/code/dansbecker/using-categorical-data-with-one-hot-encoding\ntrain_data = pd.get_dummies(train_data)\ntest_data = pd.get_dummies(test_data)\ntrain_data, test_data = train_data.align(test_data, join='outer', axis = 1)\n\ntrain_data = train_data.fillna(2)\ntest_data = test_data.fillna(2)\n\n#Imputing missing data with SimpleImputer\n#imputer = SimpleImputer()\n#imputed_train_data = imputer.fit_transform(train_data)\n#train_data = pd.DataFrame(imputed_train_data, columns=train_data.columns)\n\n#imputed_test_data = imputer.transform(test_data)\n#test_data = pd.DataFrame(imputed_test_data, columns=test_data.columns)\n#train_data, test_data = train_data.align(test_data, join='outer', axis = 1)\n\nprint(train_data.shape)\nprint(test_data.shape)\n\nprint(train_data.info())\nprint(test_data.info())\n\ndifference = (train_data.columns.difference(test_data.columns))\nprint(difference)\n\ntest_data = test_data.drop(columns=['sii'])  \ntrain_data.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:33:44.998018Z","iopub.execute_input":"2024-12-13T00:33:44.998449Z","iopub.status.idle":"2024-12-13T00:33:45.096997Z","shell.execute_reply.started":"2024-12-13T00:33:44.998402Z","shell.execute_reply":"2024-12-13T00:33:45.095864Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Logistic Regression Model","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.preprocessing import StandardScaler\n\nfrom sklearn.metrics import accuracy_score, classification_report, mean_squared_error\nfrom sklearn.model_selection import KFold, cross_val_score\n\nX = train_data.drop(columns=['sii'])  \ny = train_data['sii'] \n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.20, random_state=42)\n\n\nprint(y_train)\n\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)\n\nLR = LogisticRegression(random_state=42, max_iter=300)\n\nLR.fit(X_train_scaled, y_train)\n\ny_pred_test = LR.predict(X_test_scaled)\ny_pred_train = LR.predict(X_train_scaled)\n\nprint(\"Training accuracy:\", accuracy_score(y_train, y_pred_train))\nprint(\"Test accuracy:\", accuracy_score(y_test, y_pred_test))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:33:45.098311Z","iopub.execute_input":"2024-12-13T00:33:45.098703Z","iopub.status.idle":"2024-12-13T00:33:46.012913Z","shell.execute_reply.started":"2024-12-13T00:33:45.098671Z","shell.execute_reply":"2024-12-13T00:33:46.011760Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Gradient Boosting Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.preprocessing import StandardScaler\n\nfrom sklearn.metrics import accuracy_score, classification_report, mean_squared_error\nfrom sklearn.model_selection import KFold, cross_val_score\nX = train_data.drop(columns=['sii'])  \ny = train_data['sii'] \n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.20, random_state=42)\n\nGB = GradientBoostingClassifier(n_estimators=500, max_depth=1, min_samples_split=10, min_samples_leaf=5)\n\nGB.fit(X_train, y_train)\n\ny_pred_test = GB.predict(X_test)\ny_pred_train = GB.predict(X_train)\n\nprint(\"Training accuracy:\", accuracy_score(y_train, y_pred_train))\nprint(\"Test accuracy:\", accuracy_score(y_test, y_pred_test))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:33:46.018562Z","iopub.execute_input":"2024-12-13T00:33:46.019141Z","iopub.status.idle":"2024-12-13T00:34:01.652135Z","shell.execute_reply.started":"2024-12-13T00:33:46.019090Z","shell.execute_reply":"2024-12-13T00:34:01.651031Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# MLP Classifier","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.preprocessing import StandardScaler\n\nfrom sklearn.metrics import accuracy_score, classification_report, mean_squared_error\nfrom sklearn.model_selection import KFold, cross_val_score\nX = train_data.drop(columns=['sii'])  \ny = train_data['sii'] \n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.20, random_state=42)\n\n\nprint(y_train)\n\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)\n\nMLP = MLPClassifier(max_iter=1000, activation='logistic', hidden_layer_sizes=(3,3), alpha=0.1)\n\nMLP.fit(X_train_scaled, y_train)\n\ny_pred_test = MLP.predict(X_test_scaled)\ny_pred_train =MLP.predict(X_train_scaled)\n\nprint(\"Training accuracy:\", accuracy_score(y_train, y_pred_train))\nprint(\"Test accuracy:\", accuracy_score(y_test, y_pred_test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:34:01.653775Z","iopub.execute_input":"2024-12-13T00:34:01.654101Z","iopub.status.idle":"2024-12-13T00:34:07.601895Z","shell.execute_reply.started":"2024-12-13T00:34:01.654068Z","shell.execute_reply":"2024-12-13T00:34:07.598983Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SVC","metadata":{}},{"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.preprocessing import StandardScaler\n\nfrom sklearn.metrics import accuracy_score, classification_report, mean_squared_error\nfrom sklearn.model_selection import KFold, cross_val_score\nX = train_data.drop(columns=['sii'])  \ny = train_data['sii'] \n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.20, random_state=42)\n\n\nprint(y_train)\n\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)\n\nSVC = SVC()\n\nSVC.fit(X_train_scaled, y_train)\n\ny_pred_test = SVC.predict(X_test_scaled)\ny_pred_train =SVC.predict(X_train_scaled)\n\nprint(\"Training accuracy:\", accuracy_score(y_train, y_pred_train))\nprint(\"Test accuracy:\", accuracy_score(y_test, y_pred_test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:34:07.603157Z","iopub.execute_input":"2024-12-13T00:34:07.603622Z","iopub.status.idle":"2024-12-13T00:34:09.049316Z","shell.execute_reply.started":"2024-12-13T00:34:07.603575Z","shell.execute_reply":"2024-12-13T00:34:09.048121Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Kaggle Prediction On Test Set","metadata":{}},{"cell_type":"code","source":"X = test_data \nX_scaled = scaler.fit_transform(X)\ny_pred = LR.predict(X_scaled)\n#Predict on test data\n\n#Creating submission file\nsubmission = pd.DataFrame({\n    'id': ids,  \n    'sii': y_pred.astype(int) \n})\nprint(submission)\n\n#save to CSV\nsubmission.to_csv('submission.csv', index=False)\nprint(\"Submission file created.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-13T00:34:09.050571Z","iopub.execute_input":"2024-12-13T00:34:09.050892Z","iopub.status.idle":"2024-12-13T00:34:09.068778Z","shell.execute_reply.started":"2024-12-13T00:34:09.050862Z","shell.execute_reply":"2024-12-13T00:34:09.067762Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Parameters or sets of parameters used:\n\nLogistic Regression:\nfeatures: All features > 1700 with NaN values imputed with 2.\ndata scaled: True \nrandom_state: 42\nmax_iter: 300\n\nGradient Boosting Classifier:\nfeatures: All features > 1700 with NaN values imputed with 2.\ndata scaled: False \nn_estimators: 500\nmax_depth: 1\nmin_samples_split: 10\nmin_samples_leaf: 5\n\nMLP Classifier:\nfeatures: All features > 1700 with NaN values imputed with 2.\ndata scaled: True \nmax_iter: 1000\nactivation: logistic\nhidden_layer_sizes: (5,5)\nalpha: 0.1\n\nSVC:\nfeatures: All features > 1700 with NaN values imputed with 2.\ndata scaled: True \nmodel params: defaults\n\n\n# Changes made from the values in the first stage:  \n\nIn the first stage, my data preprocessing was made up of encoding class data, filling all NaN values with 0s, and including all the features in the training set. But for this second stage, I dropped all features that had less than 1700 examples and imputed NaN values with 2’s since the range of the sii target is 0-3, I made these changes to improve the model’s generalization on new data, and received positive results. \n\nFor my ML models, I kept the Logistic Regression model from my stage 1 results since it performed the highest out of all the models I trained on this dataset. I kept all its original parameters except max_iter which I set to 300. I made this change to prevent the model from stopping before it was finished training. In addition to the LR model, I chose three other classification models for my stage 2 results: a Gradient Boosting Classifier, an MLP Neural Network, and a Support Vector Classifier. I chose these models because I found this problem to be a classification problem and they performed well on Kaggle’s hidden test set after tuning the hyperparameters through trial and error. \n\n\n# Accuracy Estimates After Training the Models on a train_test_split With a 20% Test Size: \n\nLogistic Regression:\nTraining accuracy: 0.71\nTest accuracy: 0.70\n\nGradient Boosting Classifier:\nTraining accuracy: 0.74\nTest accuracy: 0.71\n\nMLP Classifier:\nTraining accuracy: 0.73\nTest accuracy: 0.71\n\nSVC:\nTraining accuracy: 0.76\nTest accuracy: 0.70\n","metadata":{}}]}