{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt \nimport pandas as pd \nimport plotly.express as px\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport math\nfrom plotly.subplots import make_subplots\nimport numpy as np\nfrom numpy import linalg as LA\nimport plotly.express as px\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\nimport warnings\n\nwarnings.filterwarnings('ignore')\n\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nfrom sklearn.svm import SVC\nimport numpy as np\nimport matplotlib.pyplot as plt \nimport pandas as pd \nimport plotly.express as px\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport math\nfrom plotly.subplots import make_subplots\nimport numpy as np\nfrom numpy import linalg as LA\nimport plotly.express as px\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\n\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import recall_score, f1_score, accuracy_score\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import (accuracy_score, recall_score, f1_score, \n                             classification_report, confusion_matrix)\nfrom sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.026759Z","iopub.execute_input":"2024-10-11T19:52:12.027301Z","iopub.status.idle":"2024-10-11T19:52:12.043084Z","shell.execute_reply.started":"2024-10-11T19:52:12.027231Z","shell.execute_reply":"2024-10-11T19:52:12.041799Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Exploration ","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.045507Z","iopub.execute_input":"2024-10-11T19:52:12.046011Z","iopub.status.idle":"2024-10-11T19:52:12.114610Z","shell.execute_reply.started":"2024-10-11T19:52:12.045954Z","shell.execute_reply":"2024-10-11T19:52:12.113562Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.116003Z","iopub.execute_input":"2024-10-11T19:52:12.116441Z","iopub.status.idle":"2024-10-11T19:52:12.125493Z","shell.execute_reply.started":"2024-10-11T19:52:12.116390Z","shell.execute_reply":"2024-10-11T19:52:12.124229Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.128853Z","iopub.execute_input":"2024-10-11T19:52:12.129335Z","iopub.status.idle":"2024-10-11T19:52:12.181991Z","shell.execute_reply.started":"2024-10-11T19:52:12.129277Z","shell.execute_reply":"2024-10-11T19:52:12.181093Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.183300Z","iopub.execute_input":"2024-10-11T19:52:12.183660Z","iopub.status.idle":"2024-10-11T19:52:12.209568Z","shell.execute_reply.started":"2024-10-11T19:52:12.183620Z","shell.execute_reply":"2024-10-11T19:52:12.208560Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop('id', axis=1 , inplace = True)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.210952Z","iopub.execute_input":"2024-10-11T19:52:12.211329Z","iopub.status.idle":"2024-10-11T19:52:12.218018Z","shell.execute_reply.started":"2024-10-11T19:52:12.211288Z","shell.execute_reply":"2024-10-11T19:52:12.216839Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.219712Z","iopub.execute_input":"2024-10-11T19:52:12.220171Z","iopub.status.idle":"2024-10-11T19:52:12.257011Z","shell.execute_reply.started":"2024-10-11T19:52:12.220129Z","shell.execute_reply":"2024-10-11T19:52:12.255904Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop_duplicates(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.258639Z","iopub.execute_input":"2024-10-11T19:52:12.259031Z","iopub.status.idle":"2024-10-11T19:52:12.287114Z","shell.execute_reply.started":"2024-10-11T19:52:12.258991Z","shell.execute_reply":"2024-10-11T19:52:12.285968Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.288391Z","iopub.execute_input":"2024-10-11T19:52:12.288776Z","iopub.status.idle":"2024-10-11T19:52:12.321842Z","shell.execute_reply.started":"2024-10-11T19:52:12.288708Z","shell.execute_reply":"2024-10-11T19:52:12.320964Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.326618Z","iopub.execute_input":"2024-10-11T19:52:12.326999Z","iopub.status.idle":"2024-10-11T19:52:12.537584Z","shell.execute_reply.started":"2024-10-11T19:52:12.326958Z","shell.execute_reply":"2024-10-11T19:52:12.536538Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.539638Z","iopub.execute_input":"2024-10-11T19:52:12.540132Z","iopub.status.idle":"2024-10-11T19:52:12.555199Z","shell.execute_reply.started":"2024-10-11T19:52:12.540076Z","shell.execute_reply":"2024-10-11T19:52:12.554160Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols = [col for col in data.columns if data[col].dtype != 'O']\ncat_cols = [col for col in data.columns if col not in num_cols]\nprint(f'Numerical columns: {num_cols}')\nprint(\"-------------------\")\nprint(f'Categorical columns: {cat_cols}')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.556777Z","iopub.execute_input":"2024-10-11T19:52:12.557132Z","iopub.status.idle":"2024-10-11T19:52:12.565501Z","shell.execute_reply.started":"2024-10-11T19:52:12.557092Z","shell.execute_reply":"2024-10-11T19:52:12.564424Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def detect_outliers_iqr(df):\n    outlier_indices = []\n\n    for column in df.select_dtypes(include=['float64', 'int64']).columns:\n        Q1 = df[column].quantile(0.25)\n        Q3 = df[column].quantile(0.75)\n        IQR = Q3 - Q1\n\n        lower_bound = Q1 - 1.5 * IQR\n        upper_bound = Q3 + 1.5 * IQR\n\n        outliers = df[(df[column] < lower_bound) | (df[column] > upper_bound)]\n        outlier_indices.extend(outliers.index)\n\n        print(f'Outliers in {column}:', outliers.shape[0])\n\n    return outlier_indices","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.567094Z","iopub.execute_input":"2024-10-11T19:52:12.567553Z","iopub.status.idle":"2024-10-11T19:52:12.578972Z","shell.execute_reply.started":"2024-10-11T19:52:12.567499Z","shell.execute_reply":"2024-10-11T19:52:12.577961Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"outlier_indices = detect_outliers_iqr(data)\nprint(\"Total outliers detected:\", len(set(outlier_indices)))","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.582513Z","iopub.execute_input":"2024-10-11T19:52:12.583004Z","iopub.status.idle":"2024-10-11T19:52:12.747720Z","shell.execute_reply.started":"2024-10-11T19:52:12.582962Z","shell.execute_reply":"2024-10-11T19:52:12.746797Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Analysis","metadata":{}},{"cell_type":"code","source":"data.hist(figsize=(70, 40), bins=30)  \nplt.tight_layout()  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:12.748904Z","iopub.execute_input":"2024-10-11T19:52:12.749247Z","iopub.status.idle":"2024-10-11T19:52:34.542707Z","shell.execute_reply.started":"2024-10-11T19:52:12.749201Z","shell.execute_reply":"2024-10-11T19:52:34.541489Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## How internet usage affect physical measures ","metadata":{}},{"cell_type":"markdown","source":"### Average BMI , HeartRate , Blood pressure","metadata":{}},{"cell_type":"code","source":"averages = {\n    'BMI': data['Physical-BMI'].mean(),\n    'Heart Rate': data['Physical-HeartRate'].mean(),\n    'Systolic BP': data['Physical-Systolic_BP'].mean(),\n    'Diastolic BP': data['Physical-Diastolic_BP'].mean()\n}\n\n# Convert to DataFrame for easier plotting\naverages_df = pd.DataFrame(list(averages.items()), columns=['Feature', 'Average'])\n\n# Step 2: Create a bar plot\nplt.figure(figsize=(10, 6))\nsns.barplot(x='Feature', y='Average', data=averages_df, palette='coolwarm')\nplt.title('Average Values of BMI, Heart Rate, and Blood Pressure')\nplt.ylabel('Average')\nplt.xlabel('Features')\nplt.ylim(0, averages_df['Average'].max() + 10)  # Adjusting y-axis for better visibility\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:34.544600Z","iopub.execute_input":"2024-10-11T19:52:34.545124Z","iopub.status.idle":"2024-10-11T19:52:34.841932Z","shell.execute_reply.started":"2024-10-11T19:52:34.545058Z","shell.execute_reply":"2024-10-11T19:52:34.840725Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Childern Global Assessment Scale vs internet usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.histplot(data['CGAS-CGAS_Score'], bins=30, kde=True)  \nplt.title('Distribution of Childrens Global Assessment Scale Score')\nplt.xlabel('Assessment score')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:34.843627Z","iopub.execute_input":"2024-10-11T19:52:34.844108Z","iopub.status.idle":"2024-10-11T19:52:35.253104Z","shell.execute_reply.started":"2024-10-11T19:52:34.844054Z","shell.execute_reply":"2024-10-11T19:52:35.252041Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Assuming 'data' is your DataFrame\nmean_values = data.groupby('sii')['CGAS-CGAS_Score'].mean().reset_index()\n\n# Create a bar plot\nplt.figure(figsize=(10, 8))\nsns.barplot(x='sii', y='CGAS-CGAS_Score', data=mean_values, palette='viridis')\n\n# Customize the plot\nplt.title('Average Fitness Endurance Max Stage vs Total Internet Usage')\nplt.xlabel('Total Internet Usage (hours/day)')\nplt.ylabel('Average Fitness Endurance Max Stage')\nplt.ylim(0, 100)  # Set y-axis limits from 0 to 100\nplt.grid(axis='y')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:35.254458Z","iopub.execute_input":"2024-10-11T19:52:35.254814Z","iopub.status.idle":"2024-10-11T19:52:35.554633Z","shell.execute_reply.started":"2024-10-11T19:52:35.254775Z","shell.execute_reply":"2024-10-11T19:52:35.553497Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### BMI vs SII Severity","metadata":{}},{"cell_type":"code","source":"# Plot 1: BMI vs SII Severity\nplt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-BMI', palette='pastel')\nplt.title('BMI vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('BMI')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:35.556308Z","iopub.execute_input":"2024-10-11T19:52:35.556689Z","iopub.status.idle":"2024-10-11T19:52:35.846866Z","shell.execute_reply.started":"2024-10-11T19:52:35.556648Z","shell.execute_reply":"2024-10-11T19:52:35.845791Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Blood pressure vs internet Usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-Systolic_BP', palette='pastel')\nplt.title('Blood Pressure vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('Systolic BP')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:35.848384Z","iopub.execute_input":"2024-10-11T19:52:35.848775Z","iopub.status.idle":"2024-10-11T19:52:36.169288Z","shell.execute_reply.started":"2024-10-11T19:52:35.848711Z","shell.execute_reply":"2024-10-11T19:52:36.168114Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Heart rate vs internet usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-HeartRate', palette='pastel')\nplt.title('Heart Rate vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('Heart Rate')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:36.170798Z","iopub.execute_input":"2024-10-11T19:52:36.171192Z","iopub.status.idle":"2024-10-11T19:52:36.468241Z","shell.execute_reply.started":"2024-10-11T19:52:36.171140Z","shell.execute_reply":"2024-10-11T19:52:36.467118Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Demographics influence  ","metadata":{}},{"cell_type":"markdown","source":"### Age Distribution ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.histplot(data['Basic_Demos-Age'], bins=30, kde=True)  \nplt.title('Distribution of Age Feature')\nplt.xlabel('AGE')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:36.469762Z","iopub.execute_input":"2024-10-11T19:52:36.470139Z","iopub.status.idle":"2024-10-11T19:52:36.877431Z","shell.execute_reply.started":"2024-10-11T19:52:36.470097Z","shell.execute_reply":"2024-10-11T19:52:36.876262Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Gender Distribution","metadata":{}},{"cell_type":"code","source":"# Count the number of occurrences for each sex\nGende_counts = data['Basic_Demos-Sex'].value_counts()\n\n# Create the pie chart\nplt.figure(figsize=(7, 7))\nplt.pie(Gende_counts, labels=Gende_counts.index, autopct='%1.1f%%', startangle=90, colors=sns.color_palette('pastel'))\nplt.title('Distribution of Sex in the Dataset')\nplt.axis('equal')  # Equal aspect ratio ensures that pie chart is circular\nplt.show()\n# 0-> male  1_> female","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:36.878920Z","iopub.execute_input":"2024-10-11T19:52:36.879296Z","iopub.status.idle":"2024-10-11T19:52:37.053677Z","shell.execute_reply.started":"2024-10-11T19:52:36.879255Z","shell.execute_reply":"2024-10-11T19:52:37.051924Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_severity(score):\n    if score <= 30:\n        return 'None'\n    elif score <= 49:\n        return 'Mild'\n    elif score <= 79:\n        return 'Moderate'\n    else:\n        return 'Severe'\n\ndata['Severity'] = data['PCIAT-PCIAT_Total'].apply(categorize_severity)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:37.068053Z","iopub.execute_input":"2024-10-11T19:52:37.069106Z","iopub.status.idle":"2024-10-11T19:52:37.084150Z","shell.execute_reply.started":"2024-10-11T19:52:37.069025Z","shell.execute_reply":"2024-10-11T19:52:37.083025Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(12, 6))\nsns.boxplot(x='Severity', y='Basic_Demos-Age', data=data, palette='pastel')\nplt.title('Age Distribution Across Severity Impairment Index')\nplt.xlabel('Severity Impairment Index')\nplt.ylabel('Age')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:37.085564Z","iopub.execute_input":"2024-10-11T19:52:37.086030Z","iopub.status.idle":"2024-10-11T19:52:37.429663Z","shell.execute_reply.started":"2024-10-11T19:52:37.085971Z","shell.execute_reply":"2024-10-11T19:52:37.428283Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Calculate the average age for each severity category\naverage_age = data.groupby('Severity')['Basic_Demos-Age'].mean().reset_index()\n\n# Create a line plot\nplt.figure(figsize=(12, 6))\nsns.lineplot(x='Severity', y='Basic_Demos-Age', data=average_age, marker='o')\nplt.title('Average Age Across Severity Impairment Index')\nplt.xlabel('Severity Impairment Index')\nplt.ylabel('Average Age')\nplt.xticks(rotation=45)  # Rotate x labels for better visibility\nplt.grid(True)  # Add grid for better readability\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:37.431059Z","iopub.execute_input":"2024-10-11T19:52:37.431471Z","iopub.status.idle":"2024-10-11T19:52:37.751715Z","shell.execute_reply.started":"2024-10-11T19:52:37.431427Z","shell.execute_reply":"2024-10-11T19:52:37.750517Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ninternet_usage = data.groupby(['Basic_Demos-Age', 'PreInt_EduHx-computerinternet_hoursday']).size().unstack(fill_value=0)\n\ninternet_usage = internet_usage.reset_index()\n\ninternet_usage_melted = internet_usage.melt(id_vars='Basic_Demos-Age', \n                                              var_name='Internet Usage', \n                                              value_name='Count')\nplt.figure(figsize=(12, 6))\npalette = sns.color_palette('pastel') \nsns.barplot(data=internet_usage_melted, \n            x='Basic_Demos-Age', \n            y='Count', \n            hue='Internet Usage', \n            palette=palette)\n\nplt.xlabel('Age Group', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.title('Internet Usage Distribution by Age Group', fontsize=14)\nplt.xticks(rotation=45)\n\nhandles = []\nfor i, label in enumerate(['0=Less than 1h/day', '1=Around 1h/day', '2=Around 2hs/day', '3=More than 3hs/day']):\n    handles.append(plt.Line2D([0], [0], color=palette[i], lw=4))  # Create a line for each color\n\nplt.legend(handles, \n           ['0=Less than 1h/day', '1=Around 1h/day', '2=Around 2hs/day', '3=More than 3hs/day'], \n           title='Daily Internet Usage')\n\nplt.tight_layout() \nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:37.753495Z","iopub.execute_input":"2024-10-11T19:52:37.753936Z","iopub.status.idle":"2024-10-11T19:52:38.614119Z","shell.execute_reply.started":"2024-10-11T19:52:37.753886Z","shell.execute_reply":"2024-10-11T19:52:38.613038Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sleep Quality vs internet usage","metadata":{}},{"cell_type":"code","source":"mean_sleep_time = data.groupby('PreInt_EduHx-computerinternet_hoursday')['SDS-SDS_Total_T'].mean()\n\nplt.figure(figsize=(10, 8))\nplt.plot(mean_sleep_time.index, mean_sleep_time.values, marker='o', linestyle='-', color='blue')\n\n# Customize the plot\nplt.xlabel('Internet Usage Time (hours/day)')\nplt.ylabel('Average Sleep Time (hours)')\nplt.title('Average Sleep Time vs Internet Usage Time')\n\ncustom_ticks = [ 45, 50, 55, 60, 65, 70,75]  # Define your custom tick values\nplt.yticks(custom_ticks)\n\nplt.grid(True)\n\n# Show the plot\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:38.615664Z","iopub.execute_input":"2024-10-11T19:52:38.616065Z","iopub.status.idle":"2024-10-11T19:52:38.925368Z","shell.execute_reply.started":"2024-10-11T19:52:38.616023Z","shell.execute_reply":"2024-10-11T19:52:38.924135Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Behavioural influence ","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n\nplt.figure(figsize=(12, 6))\nsns.boxplot(x='PCIAT-PCIAT_06', y='Basic_Demos-Age', data=data, palette='pastel')\nplt.title('Age Distribution for Q: How often do your childs grades suffer because of the amount of time he or she spends online?')\nplt.xlabel('Response')\nplt.ylabel('Age')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:38.926956Z","iopub.execute_input":"2024-10-11T19:52:38.927471Z","iopub.status.idle":"2024-10-11T19:52:39.298619Z","shell.execute_reply.started":"2024-10-11T19:52:38.927414Z","shell.execute_reply":"2024-10-11T19:52:39.297426Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# List of questions to analyze\nquestions = [\n    'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', \n    'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06',\n    'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09',\n    'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12',\n    'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15',\n    'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18',\n    'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20'\n]\n\nplt.figure(figsize=(15, 40))  # Adjust the size for better visibility\n\n# Create a box plot for each question\nfor i, question in enumerate(questions):\n    plt.subplot(len(questions), 1, i + 1)  # Create subplots\n    sns.boxplot(x=question, y='Basic_Demos-Age', data=data, palette='pastel')\n    plt.title(f'Age Distribution for {question}')\n    plt.xlabel('Response')\n    plt.ylabel('Age')\n\nplt.tight_layout()  # Adjust layout to prevent overlap\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:39.300422Z","iopub.execute_input":"2024-10-11T19:52:39.300822Z","iopub.status.idle":"2024-10-11T19:52:45.200936Z","shell.execute_reply.started":"2024-10-11T19:52:39.300775Z","shell.execute_reply":"2024-10-11T19:52:45.199810Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"markdown","source":"## Handling Missing Values ","metadata":{}},{"cell_type":"markdown","source":"### Drop columns with null>50","metadata":{}},{"cell_type":"code","source":"\ncol = ['CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n       'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season',\n       'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']\ncol.extend( \n        ['Physical-Waist_Circumference','Fitness_Endurance-Max_Stage',\n         'Fitness_Endurance-Time_Mins','Fitness_Endurance-Time_Sec','FGC-FGC_GSND',\n         'FGC-FGC_GSND_Zone','FGC-FGC_GSD','FGC-FGC_GSD_Zone','BIA-BIA_Activity_Level_num',\n         'BIA-BIA_BMC','BIA-BIA_BMI','BIA-BIA_BMR','BIA-BIA_DEE','BIA-BIA_ECW','BIA-BIA_FFM',\n         'BIA-BIA_FFMI','BIA-BIA_Fat','BIA-BIA_Frame_num','BIA-BIA_ICW','BIA-BIA_LDM','BIA-BIA_LST',\n         'BIA-BIA_SMM','BIA-BIA_TBW','PAQ_A-PAQ_A_Total','PAQ_C-PAQ_C_Total'])\nlen(col)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:45.202586Z","iopub.execute_input":"2024-10-11T19:52:45.203045Z","iopub.status.idle":"2024-10-11T19:52:45.214714Z","shell.execute_reply.started":"2024-10-11T19:52:45.202995Z","shell.execute_reply":"2024-10-11T19:52:45.213603Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop(col , axis =1 ,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:45.216317Z","iopub.execute_input":"2024-10-11T19:52:45.216780Z","iopub.status.idle":"2024-10-11T19:52:45.231067Z","shell.execute_reply.started":"2024-10-11T19:52:45.216705Z","shell.execute_reply":"2024-10-11T19:52:45.229873Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:45.232390Z","iopub.execute_input":"2024-10-11T19:52:45.232763Z","iopub.status.idle":"2024-10-11T19:52:45.243520Z","shell.execute_reply.started":"2024-10-11T19:52:45.232698Z","shell.execute_reply":"2024-10-11T19:52:45.242468Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned= data.copy()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:45.244883Z","iopub.execute_input":"2024-10-11T19:52:45.246080Z","iopub.status.idle":"2024-10-11T19:52:45.254579Z","shell.execute_reply.started":"2024-10-11T19:52:45.246027Z","shell.execute_reply":"2024-10-11T19:52:45.253517Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:45.257445Z","iopub.execute_input":"2024-10-11T19:52:45.258936Z","iopub.status.idle":"2024-10-11T19:52:45.266322Z","shell.execute_reply.started":"2024-10-11T19:52:45.258879Z","shell.execute_reply":"2024-10-11T19:52:45.265109Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Filling Missing Values in first 46 columns with Knn imputer","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\ndata_cleaned['Basic_Demos-Enroll_Season'] = label_encoder.fit_transform(data_cleaned['Basic_Demos-Enroll_Season'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:45.267661Z","iopub.execute_input":"2024-10-11T19:52:45.268299Z","iopub.status.idle":"2024-10-11T19:52:45.278655Z","shell.execute_reply.started":"2024-10-11T19:52:45.268258Z","shell.execute_reply":"2024-10-11T19:52:45.277648Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\nimputer = KNNImputer(n_neighbors=5)\ndata_cleaned.iloc[:, :45] = imputer.fit_transform(data_cleaned.iloc[:, :45])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:45.280177Z","iopub.execute_input":"2024-10-11T19:52:45.280626Z","iopub.status.idle":"2024-10-11T19:52:49.716602Z","shell.execute_reply.started":"2024-10-11T19:52:45.280573Z","shell.execute_reply":"2024-10-11T19:52:49.715415Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\ndata_cleaned['Severity'] = label_encoder.fit_transform(data_cleaned['Severity'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:49.718208Z","iopub.execute_input":"2024-10-11T19:52:49.718672Z","iopub.status.idle":"2024-10-11T19:52:49.726222Z","shell.execute_reply.started":"2024-10-11T19:52:49.718616Z","shell.execute_reply":"2024-10-11T19:52:49.725050Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Filling missing values in target with Kmeans","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nimport numpy as np\nimport pandas as pd\n\ndef impute_with_kmeans(df, categorical_columns, n_clusters=4):\n    # Fill missing values temporarily with the mode (or any placeholder)\n    df_temp = df.copy()\n    for col in categorical_columns:\n        df_temp[col].fillna(df_temp[col].mode()[0], inplace=True)\n\n    # Perform KMeans clustering after filling missing values\n    kmeans = KMeans(n_clusters=n_clusters, random_state=0)\n    cluster_labels = kmeans.fit_predict(df_temp)\n\n    # Impute missing values within each cluster\n    for col in categorical_columns:\n        for cluster in np.unique(cluster_labels):\n            mask = (cluster_labels == cluster) & df[col].isna()\n            most_frequent = df.loc[cluster_labels == cluster, col].mode()[0]\n            df.loc[mask, col] = most_frequent\n\n    return df\n\n# Apply KMeans-based imputation\ndata_cleaned = impute_with_kmeans(data_cleaned, ['sii'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:49.727805Z","iopub.execute_input":"2024-10-11T19:52:49.728267Z","iopub.status.idle":"2024-10-11T19:52:50.100871Z","shell.execute_reply.started":"2024-10-11T19:52:49.728214Z","shell.execute_reply":"2024-10-11T19:52:50.099070Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned['sii'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:50.102600Z","iopub.execute_input":"2024-10-11T19:52:50.103904Z","iopub.status.idle":"2024-10-11T19:52:50.114956Z","shell.execute_reply.started":"2024-10-11T19:52:50.103844Z","shell.execute_reply":"2024-10-11T19:52:50.113605Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:50.116618Z","iopub.execute_input":"2024-10-11T19:52:50.118190Z","iopub.status.idle":"2024-10-11T19:52:50.173174Z","shell.execute_reply.started":"2024-10-11T19:52:50.118128Z","shell.execute_reply":"2024-10-11T19:52:50.172024Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:50.174863Z","iopub.execute_input":"2024-10-11T19:52:50.175836Z","iopub.status.idle":"2024-10-11T19:52:50.186888Z","shell.execute_reply.started":"2024-10-11T19:52:50.175778Z","shell.execute_reply":"2024-10-11T19:52:50.185913Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handling outliers ","metadata":{}},{"cell_type":"code","source":"outlier_indices = detect_outliers_iqr(data_cleaned)\nprint(\"Total outliers detected:\", len(set(outlier_indices)))","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:50.188567Z","iopub.execute_input":"2024-10-11T19:52:50.189357Z","iopub.status.idle":"2024-10-11T19:52:50.302103Z","shell.execute_reply.started":"2024-10-11T19:52:50.189296Z","shell.execute_reply":"2024-10-11T19:52:50.300816Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for column in data_cleaned.columns:\n    plt.figure(figsize=(8, 3))  \n    sns.boxplot(data=data_cleaned[[column]], orient='h', palette=\"Set2\")\n    \n    plt.title(f'Box Plot for {column} with Outliers', fontsize=16)\n    plt.ylabel(column, fontsize=14)\n    plt.xlabel('Values', fontsize=14)\n    \n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:52:50.303715Z","iopub.execute_input":"2024-10-11T19:52:50.304160Z","iopub.status.idle":"2024-10-11T19:53:02.369339Z","shell.execute_reply.started":"2024-10-11T19:52:50.304109Z","shell.execute_reply":"2024-10-11T19:53:02.368090Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate IQR for the column 'CGAS-CGAS_Score'\nQ1 = data_cleaned['CGAS-CGAS_Score'].quantile(0.25)\nQ3 = data_cleaned['CGAS-CGAS_Score'].quantile(0.75)\nIQR = Q3 - Q1\n\n# Define outlier boundaries\nlower_bound = Q1 - 1.5 * IQR\nupper_bound = Q3 + 1.5 * IQR\n\n# Filter the entire DataFrame by keeping only rows where 'CGAS-CGAS_Score' is within bounds\ndata_copy = data_cleaned[(data_cleaned['CGAS-CGAS_Score'] >= lower_bound) & (data_cleaned['CGAS-CGAS_Score'] <= upper_bound)]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:53:02.373111Z","iopub.execute_input":"2024-10-11T19:53:02.373513Z","iopub.status.idle":"2024-10-11T19:53:02.385682Z","shell.execute_reply.started":"2024-10-11T19:53:02.373470Z","shell.execute_reply":"2024-10-11T19:53:02.384723Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"## BMI To Heart rate Ratio","metadata":{}},{"cell_type":"code","source":"data_copy['HeartRate_BMI'] = data_copy['Physical-HeartRate'] * data_copy['Physical-BMI']","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:53:02.387081Z","iopub.execute_input":"2024-10-11T19:53:02.387448Z","iopub.status.idle":"2024-10-11T19:53:02.397657Z","shell.execute_reply.started":"2024-10-11T19:53:02.387409Z","shell.execute_reply":"2024-10-11T19:53:02.396792Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Max Heart rate due to age","metadata":{}},{"cell_type":"code","source":"data_copy['HRmax'] = 220 - data_copy['Basic_Demos-Age']  # Estimate HRmax based on age","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:53:02.399149Z","iopub.execute_input":"2024-10-11T19:53:02.399494Z","iopub.status.idle":"2024-10-11T19:53:02.412281Z","shell.execute_reply.started":"2024-10-11T19:53:02.399456Z","shell.execute_reply":"2024-10-11T19:53:02.411279Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Addation Feature ","metadata":{}},{"cell_type":"code","source":"def categorize_bmi(bmi):\n    if bmi <= 18.5:\n        return 0\n    elif bmi <= 24.9:\n        return 1\n    elif bmi <= 29.9:\n        return 2\n    elif bmi <= 40:\n        return 3\n    else:\n        return 4  \n\ndata_copy['BMI_Category'] = data_copy['Physical-BMI'].apply(categorize_bmi)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:02.413759Z","iopub.execute_input":"2024-10-11T19:53:02.414295Z","iopub.status.idle":"2024-10-11T19:53:02.427607Z","shell.execute_reply.started":"2024-10-11T19:53:02.414242Z","shell.execute_reply":"2024-10-11T19:53:02.426597Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pciat_columns = [f'PCIAT-PCIAT_{i:02d}' for i in range(1, 21)]\ndata_copy['PCIAT_Time_Management'] = data_copy[pciat_columns[:5]].mean(axis=1)\ndata_copy['PCIAT_Withdrawal_Symptoms'] = data_copy[pciat_columns[5:10]].mean(axis=1)\ndata_copy['PCIAT_Neglect_Social_Life'] = data_copy[pciat_columns[10:15]].mean(axis=1)\ndata_copy['PCIAT_Lack_Control'] = data_copy[pciat_columns[15:]].mean(axis=1)\n\n\n\ndata_copy['PCIAT_mean'] = data_copy[pciat_columns].mean(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:02.431492Z","iopub.execute_input":"2024-10-11T19:53:02.431872Z","iopub.status.idle":"2024-10-11T19:53:02.454593Z","shell.execute_reply.started":"2024-10-11T19:53:02.431831Z","shell.execute_reply":"2024-10-11T19:53:02.453667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_pciat(score):\n    if score <= 20:\n         return 0\n    elif score <= 49:\n          return 1\n    elif score <= 79:\n         return 2\n    else:\n         return 3\n    \ndata_copy['PCIAT_Category'] = data_copy['PCIAT-PCIAT_Total'].apply(categorize_pciat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:02.456011Z","iopub.execute_input":"2024-10-11T19:53:02.456424Z","iopub.status.idle":"2024-10-11T19:53:02.467213Z","shell.execute_reply.started":"2024-10-11T19:53:02.456374Z","shell.execute_reply":"2024-10-11T19:53:02.466080Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_sds(score):\n    if pd.isna(score) or score < 0:\n        return np.nan\n    elif score <= 20:\n        return 0\n    elif score <= 40:\n        return 1\n    elif score <= 60:\n        return 2\n    elif score <= 80:\n        return 3\n    else:\n        return 4  \n\ndata_copy['SDS_Severity'] = data_copy['SDS-SDS_Total_Raw'].apply(categorize_sds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:02.468581Z","iopub.execute_input":"2024-10-11T19:53:02.468974Z","iopub.status.idle":"2024-10-11T19:53:02.486447Z","shell.execute_reply.started":"2024-10-11T19:53:02.468925Z","shell.execute_reply":"2024-10-11T19:53:02.485371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['Sleep_Quality_Index'] = (data_copy['SDS-SDS_Total_T'] - data_copy['SDS-SDS_Total_T'].min()) / (data_copy['SDS-SDS_Total_T'].max() - data_copy['SDS-SDS_Total_T'].min())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:02.487668Z","iopub.execute_input":"2024-10-11T19:53:02.488064Z","iopub.status.idle":"2024-10-11T19:53:02.499612Z","shell.execute_reply.started":"2024-10-11T19:53:02.488023Z","shell.execute_reply":"2024-10-11T19:53:02.498755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['Physical_Health_Index'] = ((data_copy['Physical-BMI'] - data_copy['Physical-BMI'].mean()) / data_copy['Physical-BMI'].std() + (data_copy['Physical-Systolic_BP'] - data_copy['Physical-Systolic_BP'].mean()) / data_copy['Physical-Systolic_BP'].std() + (data_copy['Physical-HeartRate'] - data_copy['Physical-HeartRate'].mean()) / data_copy['Physical-HeartRate'].std()) / 3\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:02.501014Z","iopub.execute_input":"2024-10-11T19:53:02.501455Z","iopub.status.idle":"2024-10-11T19:53:02.512826Z","shell.execute_reply.started":"2024-10-11T19:53:02.501397Z","shell.execute_reply":"2024-10-11T19:53:02.511897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['Sleep_Quality_Index'] = (data_copy['SDS-SDS_Total_T'] - data_copy['SDS-SDS_Total_T'].min()) / (data_copy['SDS-SDS_Total_T'].max() - data_copy['SDS-SDS_Total_T'].min())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:02.514126Z","iopub.execute_input":"2024-10-11T19:53:02.514474Z","iopub.status.idle":"2024-10-11T19:53:02.526140Z","shell.execute_reply.started":"2024-10-11T19:53:02.514435Z","shell.execute_reply":"2024-10-11T19:53:02.525107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fgc_columns = ['FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_TL']\n\ndata_copy['Overall_Fitness_Score'] = data_copy[fgc_columns].mean(axis=1)\ndata_copy['Internet_Usage_Score'] = data_copy['PCIAT-PCIAT_Total'] / 100\ndata_copy['Physical_Activity_Score'] = data_copy['Overall_Fitness_Score'] / data_copy['Overall_Fitness_Score'].max()\ndata_copy['Lifestyle_Score'] = ((1 - data_copy['Internet_Usage_Score']) +  data_copy['Physical_Activity_Score'] + (1 - data_copy['Sleep_Quality_Index'])) / 3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:02.527503Z","iopub.execute_input":"2024-10-11T19:53:02.527985Z","iopub.status.idle":"2024-10-11T19:53:02.543933Z","shell.execute_reply.started":"2024-10-11T19:53:02.527918Z","shell.execute_reply":"2024-10-11T19:53:02.542917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_bp(systolic, diastolic):\n    if systolic < 120 and diastolic < 80:\n        return 0\n    elif 120 <= systolic < 130 and diastolic < 80:\n        return 1\n    elif 130 <= systolic < 140 or 80 <= diastolic < 90:\n        return 2\n    else:\n        return 3 \n\ndata_copy['BP_Category'] = data_copy.apply(lambda row: categorize_bp(row['Physical-Systolic_BP'], row['Physical-Diastolic_BP']), axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:02.545282Z","iopub.execute_input":"2024-10-11T19:53:02.545594Z","iopub.status.idle":"2024-10-11T19:53:02.623062Z","shell.execute_reply.started":"2024-10-11T19:53:02.545558Z","shell.execute_reply":"2024-10-11T19:53:02.622127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['BMI_Age_Interaction'] = data_copy['Physical-BMI'] * data_copy['Basic_Demos-Age']\ndata_copy['HeartRate_BPCategory_Interaction'] = data_copy['BP_Category'] * data_copy['Basic_Demos-Age']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:02.624297Z","iopub.execute_input":"2024-10-11T19:53:02.624628Z","iopub.status.idle":"2024-10-11T19:53:02.632252Z","shell.execute_reply.started":"2024-10-11T19:53:02.624592Z","shell.execute_reply":"2024-10-11T19:53:02.631198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy = data_copy.drop(columns=['SDS-SDS_Total_T', 'SDS-SDS_Total_Raw', 'PCIAT-PCIAT_Total', 'Physical-Height', 'Physical-Weight', 'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_TL', 'FGC-FGC_CU_Zone', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_TL_Zone', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20'])\n\n\n\ndata_copy = data_copy.drop(columns=['PCIAT_Time_Management', 'PCIAT_Withdrawal_Symptoms', 'PCIAT_Neglect_Social_Life', 'PCIAT_Lack_Control',  'Basic_Demos-Age', 'Physical-BMI'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:02.633480Z","iopub.execute_input":"2024-10-11T19:53:02.633847Z","iopub.status.idle":"2024-10-11T19:53:02.648174Z","shell.execute_reply.started":"2024-10-11T19:53:02.633807Z","shell.execute_reply":"2024-10-11T19:53:02.647261Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Body Strength & Flexibility","metadata":{}},{"cell_type":"code","source":"# data_copy['BodyStrength&Flexibility'] = data_copy['FGC-FGC_CU'] + data_copy['FGC-FGC_PU'] + data_copy['FGC-FGC_SRL'] + data_copy['FGC-FGC_SRR'] + data_copy['FGC-FGC_TL']\n# data_copy['BodyStrength&Flexibility_class'] = data_copy['FGC-FGC_CU_Zone'] + data_copy['FGC-FGC_PU_Zone'] + data_copy['FGC-FGC_SRL_Zone'] + data_copy['FGC-FGC_SRR_Zone'] + data_copy['FGC-FGC_TL_Zone']\n\n# data_copy = data_copy.drop(['FGC-FGC_CU','FGC-FGC_PU','FGC-FGC_SRL','FGC-FGC_SRR','FGC-FGC_TL'], axis=1)\n\n# data_copy = data_copy.drop(['FGC-FGC_CU_Zone','FGC-FGC_PU_Zone','FGC-FGC_SRL_Zone','FGC-FGC_SRR_Zone','FGC-FGC_TL_Zone'], axis=1)                          \n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:53:02.649576Z","iopub.execute_input":"2024-10-11T19:53:02.649940Z","iopub.status.idle":"2024-10-11T19:53:02.661241Z","shell.execute_reply.started":"2024-10-11T19:53:02.649896Z","shell.execute_reply":"2024-10-11T19:53:02.660164Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:53:02.662447Z","iopub.execute_input":"2024-10-11T19:53:02.662851Z","iopub.status.idle":"2024-10-11T19:53:02.677681Z","shell.execute_reply.started":"2024-10-11T19:53:02.662802Z","shell.execute_reply":"2024-10-11T19:53:02.676636Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Transfomation and scaling","metadata":{}},{"cell_type":"markdown","source":"## Data Splitting ","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = data_copy.drop('sii', axis=1)  \ny = data_copy['sii']  \nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)\nprint(X_train.shape, X_test.shape, y_train.shape, y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:53:02.678989Z","iopub.execute_input":"2024-10-11T19:53:02.679322Z","iopub.status.idle":"2024-10-11T19:53:02.697014Z","shell.execute_reply.started":"2024-10-11T19:53:02.679284Z","shell.execute_reply":"2024-10-11T19:53:02.696002Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"value_counts = data_copy['sii'].value_counts()\n\n# Plotting the value counts\nplt.figure(figsize=(8, 6))\nvalue_counts.plot(kind='bar', color='skyblue')\nplt.title(\"Value Counts of 'sii'\")\nplt.xlabel('Categories')\nplt.ylabel('Counts')\nplt.xticks(rotation=0)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:02.698374Z","iopub.execute_input":"2024-10-11T19:53:02.698704Z","iopub.status.idle":"2024-10-11T19:53:02.915589Z","shell.execute_reply.started":"2024-10-11T19:53:02.698667Z","shell.execute_reply":"2024-10-11T19:53:02.914456Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handling imbalanced data ","metadata":{}},{"cell_type":"code","source":"from collections import Counter\nfrom imblearn.over_sampling import SMOTE\n\n# Check the class distribution before applying SMOTE\nclass_distribution_before = Counter(y_train)\nprint(\"Class distribution before SMOTE:\", class_distribution_before)\n\n# Initialize SMOTE with specified parameters\nsmote = SMOTE(sampling_strategy='auto', random_state=42)\n\n# Apply SMOTE to the training data\nX_train_resampled, y_train_resampled = smote.fit_resample(X_train, y_train)\n\n# Check the class distribution after applying SMOTE\nclass_distribution_after = Counter(y_train_resampled)\nprint(\"Class distribution after SMOTE:\", class_distribution_after)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:53:02.916885Z","iopub.execute_input":"2024-10-11T19:53:02.917246Z","iopub.status.idle":"2024-10-11T19:53:02.951128Z","shell.execute_reply.started":"2024-10-11T19:53:02.917207Z","shell.execute_reply":"2024-10-11T19:53:02.950018Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train = y_train_resampled","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:02.952366Z","iopub.execute_input":"2024-10-11T19:53:02.952689Z","iopub.status.idle":"2024-10-11T19:53:02.960834Z","shell.execute_reply.started":"2024-10-11T19:53:02.952649Z","shell.execute_reply":"2024-10-11T19:53:02.959797Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Scaling","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import RobustScaler\nscaler = RobustScaler()  # or StandardScaler()\n\n# Fit the scaler on X_train and transform both X_train and X_test\nX_train = pd.DataFrame(scaler.fit_transform(X_train_resampled), columns=X_train_resampled.columns)\nX_test = pd.DataFrame(scaler.transform(X_test), columns=X_test.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:53:02.962206Z","iopub.execute_input":"2024-10-11T19:53:02.962591Z","iopub.status.idle":"2024-10-11T19:53:02.992699Z","shell.execute_reply.started":"2024-10-11T19:53:02.962552Z","shell.execute_reply":"2024-10-11T19:53:02.991801Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.shape\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:53:03.003504Z","iopub.execute_input":"2024-10-11T19:53:03.003926Z","iopub.status.idle":"2024-10-11T19:53:03.011016Z","shell.execute_reply.started":"2024-10-11T19:53:03.003883Z","shell.execute_reply":"2024-10-11T19:53:03.009904Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:53:03.012459Z","iopub.execute_input":"2024-10-11T19:53:03.012918Z","iopub.status.idle":"2024-10-11T19:53:03.022195Z","shell.execute_reply.started":"2024-10-11T19:53:03.012865Z","shell.execute_reply":"2024-10-11T19:53:03.021135Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PCA","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA \npca = PCA(n_components= 20 )  # Reduce to 2 components\nX_train = pca.fit_transform(X_train)\nX_test = pca.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:03.023571Z","iopub.execute_input":"2024-10-11T19:53:03.024034Z","iopub.status.idle":"2024-10-11T19:53:03.056086Z","shell.execute_reply.started":"2024-10-11T19:53:03.023981Z","shell.execute_reply":"2024-10-11T19:53:03.054797Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modeling ","metadata":{}},{"cell_type":"code","source":"# pip install lazypredict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:03.058028Z","iopub.execute_input":"2024-10-11T19:53:03.058977Z","iopub.status.idle":"2024-10-11T19:53:03.064353Z","shell.execute_reply.started":"2024-10-11T19:53:03.058888Z","shell.execute_reply":"2024-10-11T19:53:03.062997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from lazypredict.Supervised import LazyClassifier\n# from sklearn.model_selection import train_test_split\n# from sklearn.datasets import load_iris\n# import pandas as pd\n\n\n\n# clf = LazyClassifier(verbose=0, ignore_warnings=True, custom_metric=None)\n\n# models, predictions = clf.fit(X_train, X_test, y_train, y_test)\n\n# print(models)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:03.065883Z","iopub.execute_input":"2024-10-11T19:53:03.066439Z","iopub.status.idle":"2024-10-11T19:53:03.082074Z","shell.execute_reply.started":"2024-10-11T19:53:03.066365Z","shell.execute_reply":"2024-10-11T19:53:03.080978Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result1, result2, result3 = [], [], [] ","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:53:03.083853Z","iopub.execute_input":"2024-10-11T19:53:03.084545Z","iopub.status.idle":"2024-10-11T19:53:03.090630Z","shell.execute_reply.started":"2024-10-11T19:53:03.084474Z","shell.execute_reply":"2024-10-11T19:53:03.089445Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_f1_scores = []\ntest_f1_scores = []\nmodel_names = []\n\ndef modeling(model, model_name):\n    # Fit the model using the scaled resampled training data\n    global train_f1_scores, test_f1_scores, model_names\n\n    model.fit(X_train, y_train_resampled)  # Use y_train_resampled here\n    \n    # Predictions\n    train_pred = model.predict(X_train)    # Predictions on the scaled resampled train set\n    test_pred = model.predict(X_test)      # Predictions on the scaled original test set\n    \n    # Calculate metrics with specified average\n    train_accuracy = accuracy_score(y_train_resampled, train_pred) * 100\n    train_recall = recall_score(y_train_resampled, train_pred, average='weighted') * 100\n    train_f1_score = f1_score(y_train_resampled, train_pred, average='weighted') * 100\n    \n    test_accuracy = accuracy_score(y_test, test_pred) * 100\n    test_recall = recall_score(y_test, test_pred, average='weighted') * 100\n    test_f1_score = f1_score(y_test, test_pred, average='weighted') * 100\n\n\n    train_f1_scores.append(train_f1_score)\n    test_f1_scores.append(test_f1_score)\n    model_names.append(model_name)\n\n    \n    # Append results\n    result1.append(test_accuracy)\n    result2.append(test_recall)\n    result3.append(test_f1_score)\n    \n    print(\"Classification Report for Test Data:\")\n    print(classification_report(y_test, test_pred))\n    \n    print(\"\\nClassification Report for Scaled Resampled Train Data:\")\n    print(classification_report(y_train_resampled, train_pred))  # Use y_train_resampled here\n    \n    # Accuracy, Recall, and F1 Scores\n    print(f'Training Accuracy: {train_accuracy}, Train Recall: {train_recall}, Train F1: {train_f1_score}')\n    print(f'Test Accuracy: {test_accuracy}, Test Recall: {test_recall}, Test F1: {test_f1_score}')\n    \n    # Confusion matrix\n    cm = confusion_matrix(y_test, test_pred)\n    sns.heatmap(cm, annot=True, fmt='0.2f', cmap='YlGnBu', linewidths=1)\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n    plt.title('Confusion Matrix')\n    plt.show()\n\n    \n\n    labels = ['Train F1 Score', 'Test F1 Score']\n    scores = [train_f1_score, test_f1_score]\n\n    plt.figure(figsize=(4, 7))  # Larger figure for better design\n    bars = plt.bar(labels, scores, color=['#1f77b4', '#ff7f0e'], edgecolor='black')\n\n    plt.ylim(0, 100)  # Since F1 scores are percentages, 0 to 100 is the range\n    plt.title(f'F1 Score Comparison: {model_name}', fontsize=14)\n    plt.ylabel('F1 Score (%)')\n    \n    # Display values in the center of the bars\n    for bar in bars:\n        height = bar.get_height()\n        plt.text(bar.get_x() + bar.get_width()/2., height/2, f'{height:.2f}%', \n                 ha='center', va='center', color='white', fontsize=12)\n\n    plt.grid(axis='y', linestyle='--', alpha=0.7)  # Add grid lines for better readability\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:55:39.272345Z","iopub.execute_input":"2024-10-11T19:55:39.273431Z","iopub.status.idle":"2024-10-11T19:55:39.290112Z","shell.execute_reply.started":"2024-10-11T19:55:39.273378Z","shell.execute_reply":"2024-10-11T19:55:39.288979Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression  \nlogistic_regression = LogisticRegression()\nmodeling(logistic_regression, 'Logistic Regression')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:55:41.600299Z","iopub.execute_input":"2024-10-11T19:55:41.600721Z","iopub.status.idle":"2024-10-11T19:55:42.617941Z","shell.execute_reply.started":"2024-10-11T19:55:41.600679Z","shell.execute_reply":"2024-10-11T19:55:42.616702Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SVM = SVC()\nmodeling(SVM, 'SVM')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:56:11.473839Z","iopub.execute_input":"2024-10-11T19:56:11.474985Z","iopub.status.idle":"2024-10-11T19:56:12.945516Z","shell.execute_reply.started":"2024-10-11T19:56:11.474922Z","shell.execute_reply":"2024-10-11T19:56:12.944326Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decision_tree_classifier = DecisionTreeClassifier()\nmodeling(decision_tree_classifier, 'Decision Tree')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:57:09.241434Z","iopub.execute_input":"2024-10-11T19:57:09.242332Z","iopub.status.idle":"2024-10-11T19:57:10.132563Z","shell.execute_reply.started":"2024-10-11T19:57:09.242277Z","shell.execute_reply":"2024-10-11T19:57:10.131345Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decision_tree_classifier = DecisionTreeClassifier( max_depth= 7 , min_samples_split= 7)\nmodeling(decision_tree_classifier, 'Decision Tree Tuned')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:57:56.874918Z","iopub.execute_input":"2024-10-11T19:57:56.875398Z","iopub.status.idle":"2024-10-11T19:57:57.685633Z","shell.execute_reply.started":"2024-10-11T19:57:56.875350Z","shell.execute_reply":"2024-10-11T19:57:57.684594Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sgd_classifier = SGDClassifier(max_iter = 500) \nmodeling(sgd_classifier, 'SGD')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:58:03.918621Z","iopub.execute_input":"2024-10-11T19:58:03.919096Z","iopub.status.idle":"2024-10-11T19:58:04.721609Z","shell.execute_reply.started":"2024-10-11T19:58:03.919049Z","shell.execute_reply":"2024-10-11T19:58:04.720517Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nknn = KNeighborsClassifier()\nmodeling(knn, 'KNN Classifier')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:58:39.933402Z","iopub.execute_input":"2024-10-11T19:58:39.933880Z","iopub.status.idle":"2024-10-11T19:58:41.486156Z","shell.execute_reply.started":"2024-10-11T19:58:39.933830Z","shell.execute_reply":"2024-10-11T19:58:41.485138Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n\n# Create an instance of GaussianNB\nnaive_bayes = GaussianNB()\n\n# Call your modeling function\nmodeling(naive_bayes, 'Naive Bayes')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:59:01.627479Z","iopub.execute_input":"2024-10-11T19:59:01.627964Z","iopub.status.idle":"2024-10-11T19:59:02.220951Z","shell.execute_reply.started":"2024-10-11T19:59:01.627915Z","shell.execute_reply":"2024-10-11T19:59:02.219813Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nrandom_forest = RandomForestClassifier(random_state=1)\nmodeling(random_forest, 'Random Forest')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:59:05.565964Z","iopub.execute_input":"2024-10-11T19:59:05.566398Z","iopub.status.idle":"2024-10-11T19:59:09.017439Z","shell.execute_reply.started":"2024-10-11T19:59:05.566355Z","shell.execute_reply":"2024-10-11T19:59:09.016305Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"adaBosster=AdaBoostClassifier( n_estimators=50)\nmodeling(adaBosster, 'AdaBoost')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:59:13.791067Z","iopub.execute_input":"2024-10-11T19:59:13.791510Z","iopub.status.idle":"2024-10-11T19:59:15.912365Z","shell.execute_reply.started":"2024-10-11T19:59:13.791466Z","shell.execute_reply":"2024-10-11T19:59:15.911245Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingClassifier\n\ngradient=GradientBoostingClassifier()\nmodeling(gradient, 'Gradient Boost')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:59:17.774847Z","iopub.execute_input":"2024-10-11T19:59:17.775779Z","iopub.status.idle":"2024-10-11T19:59:40.031276Z","shell.execute_reply.started":"2024-10-11T19:59:17.775708Z","shell.execute_reply":"2024-10-11T19:59:40.030159Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_f1_scores():\n    plt.figure(figsize=(12, 6))\n    x = range(len(model_names))\n    \n    plt.plot(x, train_f1_scores, 'bo-', label='Training F1 Score')\n    plt.plot(x, test_f1_scores, 'ro-', label='Testing F1 Score')\n    \n    plt.xlabel('Models')\n    plt.ylabel('F1 Score (Weighted)')\n    plt.title('Training and Testing F1 Scores for Different Models')\n    plt.xticks(x, model_names, rotation=45, ha='right')\n    plt.legend()\n    plt.tight_layout()\n    plt.show()\n\ndef plot_f1_scores1():\n    plt.figure(figsize=(12, 6))\n    \n    x = np.arange(len(model_names))  # the label locations\n    width = 0.35  # the width of the bars\n    \n    # Create the bars\n    rects1 = plt.bar(x - width/2, train_f1_scores, width, label='Train', color='blue', alpha=0.7)\n    rects2 = plt.bar(x + width/2, test_f1_scores, width, label='Test', color='red', alpha=0.7)\n\n    # Add some text for labels, title and custom x-axis tick labels, etc.\n    plt.ylabel('F1 Score (Weighted)')\n    plt.title('F1 Scores for Different Models (Training and Testing)')\n    plt.xticks(x, model_names, rotation=45, ha='right')\n    plt.legend()\n\n    # Add value labels on the bars\n    def autolabel(rects):\n        for rect in rects:\n            height = rect.get_height()\n            plt.annotate(f'{height:.1f}',\n                        xy=(rect.get_x() + rect.get_width() / 2, height),\n                        xytext=(0, 3),  # 3 points vertical offset\n                        textcoords=\"offset points\",\n                        ha='center', va='bottom')\n\n    autolabel(rects1)\n    autolabel(rects2)\n\n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:38.315226Z","iopub.execute_input":"2024-10-11T19:53:38.315572Z","iopub.status.idle":"2024-10-11T19:53:38.329365Z","shell.execute_reply.started":"2024-10-11T19:53:38.315534Z","shell.execute_reply":"2024-10-11T19:53:38.328011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_f1_scores()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:38.331039Z","iopub.execute_input":"2024-10-11T19:53:38.331429Z","iopub.status.idle":"2024-10-11T19:53:38.778888Z","shell.execute_reply.started":"2024-10-11T19:53:38.331386Z","shell.execute_reply":"2024-10-11T19:53:38.777704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_f1_scores1()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:53:38.780445Z","iopub.execute_input":"2024-10-11T19:53:38.780826Z","iopub.status.idle":"2024-10-11T19:53:39.378654Z","shell.execute_reply.started":"2024-10-11T19:53:38.780788Z","shell.execute_reply":"2024-10-11T19:53:39.377469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}