{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Installing necessary packages","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report, confusion_matrix\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:18:25.631375Z","iopub.execute_input":"2024-09-25T12:18:25.631810Z","iopub.status.idle":"2024-09-25T12:18:29.935750Z","shell.execute_reply.started":"2024-09-25T12:18:25.631766Z","shell.execute_reply":"2024-09-25T12:18:29.933793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> # Data Loading and Preprocessing","metadata":{}},{"cell_type":"code","source":"# Load the CSV data\ntrain_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ntest_data = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:18:51.668770Z","iopub.execute_input":"2024-09-25T12:18:51.669420Z","iopub.status.idle":"2024-09-25T12:18:51.828707Z","shell.execute_reply.started":"2024-09-25T12:18:51.669372Z","shell.execute_reply":"2024-09-25T12:18:51.827209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get Statistical details\ntrain_data.describe().transpose()","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:19:03.461015Z","iopub.execute_input":"2024-09-25T12:19:03.461465Z","iopub.status.idle":"2024-09-25T12:19:03.678359Z","shell.execute_reply.started":"2024-09-25T12:19:03.461420Z","shell.execute_reply":"2024-09-25T12:19:03.676698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:19:11.983891Z","iopub.execute_input":"2024-09-25T12:19:11.984422Z","iopub.status.idle":"2024-09-25T12:19:12.021312Z","shell.execute_reply.started":"2024-09-25T12:19:11.984374Z","shell.execute_reply":"2024-09-25T12:19:12.020016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution of Label Data\ntrain_data['sii'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:19:23.077602Z","iopub.execute_input":"2024-09-25T12:19:23.078552Z","iopub.status.idle":"2024-09-25T12:19:23.090903Z","shell.execute_reply.started":"2024-09-25T12:19:23.078499Z","shell.execute_reply":"2024-09-25T12:19:23.089620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Handling Missing Values","metadata":{}},{"cell_type":"code","source":"# We'll start by selecting columns with more than 50% non-null values and filling missing values\nthreshold = 0.5 * len(train_data)\ncolumns_with_data = train_data.columns[train_data.isnull().sum() < threshold]\ntrain_data = train_data[columns_with_data]\n# Replace all missing values with 0\ntrain_data = train_data.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:19:32.737794Z","iopub.execute_input":"2024-09-25T12:19:32.738978Z","iopub.status.idle":"2024-09-25T12:19:32.768651Z","shell.execute_reply.started":"2024-09-25T12:19:32.738903Z","shell.execute_reply":"2024-09-25T12:19:32.767485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define the target column\ntarget_column = 'sii'\n# Remove rows where the target column 'sii' is NaN (if there are any)\ntrain_data_cleaned = train_data.dropna(subset=[target_column])\n# Check the results\ntrain_data_cleaned.head()\ntrain_data_cleaned.info()","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:19:59.987722Z","iopub.execute_input":"2024-09-25T12:19:59.988191Z","iopub.status.idle":"2024-09-25T12:20:00.016207Z","shell.execute_reply.started":"2024-09-25T12:19:59.988148Z","shell.execute_reply":"2024-09-25T12:20:00.015026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA (Exploratory Data Analysis)","metadata":{}},{"cell_type":"code","source":"# Categorical columns in the dataset\ncategorical_columns = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', \n                       'FGC-Season', 'BIA-Season', 'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']\n# Plotting boxplots for 'sii' against each categorical column\nplt.figure(figsize=(16, 24))\nfor i, col in enumerate(categorical_columns, 1):\n    plt.subplot(4, 2, i)  # 4 rows, 2 columns, plot i\n    sns.boxplot(x=col, y='sii', data=train_data_cleaned)\n    plt.xticks(rotation=45)\n    plt.title(f\"'sii' vs {col}\")\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:20:27.919507Z","iopub.execute_input":"2024-09-25T12:20:27.919991Z","iopub.status.idle":"2024-09-25T12:20:30.580418Z","shell.execute_reply.started":"2024-09-25T12:20:27.919915Z","shell.execute_reply":"2024-09-25T12:20:30.579125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot target column 'sii' with numerical columns\nnumerical_cols = train_data_cleaned.select_dtypes(include=['float64', 'int64']).columns\n# Set the number of plots per row\nplots_per_row = 5\nn_rows = (len(numerical_cols) + plots_per_row - 1) // plots_per_row\nplt.figure(figsize=(20, 4 * n_rows))\nfor i, col in enumerate(numerical_cols):\n    plt.subplot(n_rows, plots_per_row, i + 1)\n    sns.boxplot(x='sii', y=col, data=train_data_cleaned)\n    plt.title(col)\n    plt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:20:41.307135Z","iopub.execute_input":"2024-09-25T12:20:41.307602Z","iopub.status.idle":"2024-09-25T12:21:48.121445Z","shell.execute_reply.started":"2024-09-25T12:20:41.307550Z","shell.execute_reply":"2024-09-25T12:21:48.120085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Encoding categorical columns","metadata":{}},{"cell_type":"code","source":"# Identify categorical columns for seasons\nseason_cols = [\n    'Basic_Demos-Enroll_Season', \n    'CGAS-Season', \n    'Physical-Season', \n    'FGC-Season', \n    'BIA-Season', \n    'PCIAT-Season', \n    'SDS-Season', \n    'PreInt_EduHx-Season'\n]\n# Create a mapping dictionary for seasons\nseason_mapping = {\n    'Spring': 0,\n    'Summer': 1,\n    'Fall': 2,\n    'Winter': 3\n}\n# Apply manual encoding to the categorical columns\nfor col in season_cols:\n    if col in train_data_cleaned.columns:\n        train_data_cleaned[col] = train_data_cleaned[col].replace(season_mapping)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:21:50.025506Z","iopub.execute_input":"2024-09-25T12:21:50.025998Z","iopub.status.idle":"2024-09-25T12:21:50.075879Z","shell.execute_reply.started":"2024-09-25T12:21:50.025952Z","shell.execute_reply":"2024-09-25T12:21:50.074431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Correlation Matrix","metadata":{}},{"cell_type":"code","source":"# Drop the 'id' column if present\ntrain_data_no_id = train_data_cleaned.drop(columns=['id'], errors='ignore')\n# Calculate the correlation matrix\ncorrelation_matrix = train_data_no_id.corr()\n# Plot the heatmap\nplt.figure(figsize=(30, 30))\nsns.heatmap(correlation_matrix, annot=True, fmt='.1f', cmap='coolwarm', square=True)\nplt.title('Correlation Heatmap')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:21:59.929341Z","iopub.execute_input":"2024-09-25T12:21:59.929777Z","iopub.status.idle":"2024-09-25T12:22:11.369894Z","shell.execute_reply.started":"2024-09-25T12:21:59.929738Z","shell.execute_reply":"2024-09-25T12:22:11.368203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# We will select columns according to Correlation. We will not include columns with 0 correlation for the training","metadata":{}},{"cell_type":"code","source":"X = train_data_cleaned.drop(columns=['id', 'sii', 'Basic_Demos-Enroll_Season', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR_Zone','BIA-BIA_BMC', 'BIA-BIA_Fat'])\ny = train_data_cleaned['sii']","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:41:47.743094Z","iopub.execute_input":"2024-09-25T12:41:47.743730Z","iopub.status.idle":"2024-09-25T12:41:47.757230Z","shell.execute_reply.started":"2024-09-25T12:41:47.743674Z","shell.execute_reply":"2024-09-25T12:41:47.755383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=2)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:42:22.468022Z","iopub.execute_input":"2024-09-25T12:42:22.468542Z","iopub.status.idle":"2024-09-25T12:42:22.482615Z","shell.execute_reply.started":"2024-09-25T12:42:22.468496Z","shell.execute_reply":"2024-09-25T12:42:22.481050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Feature Scaling\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:42:28.476462Z","iopub.execute_input":"2024-09-25T12:42:28.476910Z","iopub.status.idle":"2024-09-25T12:42:28.508500Z","shell.execute_reply.started":"2024-09-25T12:42:28.476870Z","shell.execute_reply":"2024-09-25T12:42:28.506237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply PCA\npca = PCA(n_components=0.95)  # Keep 95% of the variance\nX_train_pca = pca.fit_transform(X_train_scaled)\nX_test_pca = pca.transform(X_test_scaled)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:42:35.181343Z","iopub.execute_input":"2024-09-25T12:42:35.181826Z","iopub.status.idle":"2024-09-25T12:42:35.244208Z","shell.execute_reply.started":"2024-09-25T12:42:35.181779Z","shell.execute_reply":"2024-09-25T12:42:35.241901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training the Random Forest Classifier Model","metadata":{}},{"cell_type":"code","source":"# Instantiate the Model\nrf_model = RandomForestClassifier(n_estimators=100, random_state=2)\n# Fit the Model on PCA-transformed data\nrf_model.fit(X_train_pca, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:43:23.633722Z","iopub.execute_input":"2024-09-25T12:43:23.635175Z","iopub.status.idle":"2024-09-25T12:43:25.436366Z","shell.execute_reply.started":"2024-09-25T12:43:23.635117Z","shell.execute_reply":"2024-09-25T12:43:25.435020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Evaluation","metadata":{}},{"cell_type":"code","source":"# Make predictions on the test set\ny_pred = rf_model.predict(X_test_pca)\n# Generate a classification report\nprint(\"Classification Report:\")\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:43:51.568202Z","iopub.execute_input":"2024-09-25T12:43:51.568762Z","iopub.status.idle":"2024-09-25T12:43:51.618836Z","shell.execute_reply.started":"2024-09-25T12:43:51.568711Z","shell.execute_reply":"2024-09-25T12:43:51.617291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate a confusion matrix\nprint(\"Confusion Matrix:\")\nprint(confusion_matrix(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:44:06.027743Z","iopub.execute_input":"2024-09-25T12:44:06.029107Z","iopub.status.idle":"2024-09-25T12:44:06.042856Z","shell.execute_reply.started":"2024-09-25T12:44:06.029044Z","shell.execute_reply":"2024-09-25T12:44:06.040907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the accuracy on the test set\naccuracy = rf_model.score(X_test_pca, y_test)\nprint(f\"Model Accuracy: {accuracy}\")","metadata":{"execution":{"iopub.status.busy":"2024-09-25T12:44:15.785917Z","iopub.execute_input":"2024-09-25T12:44:15.786512Z","iopub.status.idle":"2024-09-25T12:44:15.827541Z","shell.execute_reply.started":"2024-09-25T12:44:15.786454Z","shell.execute_reply":"2024-09-25T12:44:15.826026Z"},"trusted":true},"execution_count":null,"outputs":[]}]}