{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt \nimport pandas as pd \nimport plotly.express as px\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport math\nfrom plotly.subplots import make_subplots\nimport numpy as np\nfrom numpy import linalg as LA\nimport plotly.express as px\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\nimport warnings\n\nwarnings.filterwarnings('ignore')\n\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nfrom sklearn.svm import SVC\nimport numpy as np\nimport matplotlib.pyplot as plt \nimport pandas as pd \nimport plotly.express as px\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport math\nfrom plotly.subplots import make_subplots\nimport numpy as np\nfrom numpy import linalg as LA\nimport plotly.express as px\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\n\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import recall_score, f1_score, accuracy_score\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import (accuracy_score, recall_score, f1_score, \n                             classification_report, confusion_matrix)\nfrom sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:55.531862Z","iopub.execute_input":"2024-10-11T19:34:55.532323Z","iopub.status.idle":"2024-10-11T19:34:55.545173Z","shell.execute_reply.started":"2024-10-11T19:34:55.532275Z","shell.execute_reply":"2024-10-11T19:34:55.544076Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Exploration ","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:55.563108Z","iopub.execute_input":"2024-10-11T19:34:55.563575Z","iopub.status.idle":"2024-10-11T19:34:55.637136Z","shell.execute_reply.started":"2024-10-11T19:34:55.563527Z","shell.execute_reply":"2024-10-11T19:34:55.636133Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:55.639112Z","iopub.execute_input":"2024-10-11T19:34:55.639498Z","iopub.status.idle":"2024-10-11T19:34:55.651044Z","shell.execute_reply.started":"2024-10-11T19:34:55.639452Z","shell.execute_reply":"2024-10-11T19:34:55.649822Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:55.831916Z","iopub.execute_input":"2024-10-11T19:34:55.832353Z","iopub.status.idle":"2024-10-11T19:34:55.890362Z","shell.execute_reply.started":"2024-10-11T19:34:55.832309Z","shell.execute_reply":"2024-10-11T19:34:55.887610Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:55.892464Z","iopub.execute_input":"2024-10-11T19:34:55.892881Z","iopub.status.idle":"2024-10-11T19:34:55.922938Z","shell.execute_reply.started":"2024-10-11T19:34:55.892834Z","shell.execute_reply":"2024-10-11T19:34:55.921671Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop('id', axis=1 , inplace = True)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:55.927246Z","iopub.execute_input":"2024-10-11T19:34:55.927744Z","iopub.status.idle":"2024-10-11T19:34:55.935385Z","shell.execute_reply.started":"2024-10-11T19:34:55.927697Z","shell.execute_reply":"2024-10-11T19:34:55.934257Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:56.055681Z","iopub.execute_input":"2024-10-11T19:34:56.056542Z","iopub.status.idle":"2024-10-11T19:34:56.091321Z","shell.execute_reply.started":"2024-10-11T19:34:56.056491Z","shell.execute_reply":"2024-10-11T19:34:56.090256Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop_duplicates(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:56.131545Z","iopub.execute_input":"2024-10-11T19:34:56.132682Z","iopub.status.idle":"2024-10-11T19:34:56.161743Z","shell.execute_reply.started":"2024-10-11T19:34:56.132613Z","shell.execute_reply":"2024-10-11T19:34:56.160797Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:56.231834Z","iopub.execute_input":"2024-10-11T19:34:56.232762Z","iopub.status.idle":"2024-10-11T19:34:56.264922Z","shell.execute_reply.started":"2024-10-11T19:34:56.232704Z","shell.execute_reply":"2024-10-11T19:34:56.263818Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:56.363994Z","iopub.execute_input":"2024-10-11T19:34:56.364900Z","iopub.status.idle":"2024-10-11T19:34:56.564225Z","shell.execute_reply.started":"2024-10-11T19:34:56.364853Z","shell.execute_reply":"2024-10-11T19:34:56.563152Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:56.566250Z","iopub.execute_input":"2024-10-11T19:34:56.566626Z","iopub.status.idle":"2024-10-11T19:34:56.583168Z","shell.execute_reply.started":"2024-10-11T19:34:56.566582Z","shell.execute_reply":"2024-10-11T19:34:56.581980Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols = [col for col in data.columns if data[col].dtype != 'O']\ncat_cols = [col for col in data.columns if col not in num_cols]\nprint(f'Numerical columns: {num_cols}')\nprint(\"-------------------\")\nprint(f'Categorical columns: {cat_cols}')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:56.584521Z","iopub.execute_input":"2024-10-11T19:34:56.584926Z","iopub.status.idle":"2024-10-11T19:34:56.597966Z","shell.execute_reply.started":"2024-10-11T19:34:56.584882Z","shell.execute_reply":"2024-10-11T19:34:56.596844Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def detect_outliers_iqr(df):\n    outlier_indices = []\n\n    for column in df.select_dtypes(include=['float64', 'int64']).columns:\n        Q1 = df[column].quantile(0.25)\n        Q3 = df[column].quantile(0.75)\n        IQR = Q3 - Q1\n\n        lower_bound = Q1 - 1.5 * IQR\n        upper_bound = Q3 + 1.5 * IQR\n\n        outliers = df[(df[column] < lower_bound) | (df[column] > upper_bound)]\n        outlier_indices.extend(outliers.index)\n\n        print(f'Outliers in {column}:', outliers.shape[0])\n\n    return outlier_indices","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:56.631605Z","iopub.execute_input":"2024-10-11T19:34:56.632607Z","iopub.status.idle":"2024-10-11T19:34:56.643768Z","shell.execute_reply.started":"2024-10-11T19:34:56.632557Z","shell.execute_reply":"2024-10-11T19:34:56.642494Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"outlier_indices = detect_outliers_iqr(data)\nprint(\"Total outliers detected:\", len(set(outlier_indices)))","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:56.677794Z","iopub.execute_input":"2024-10-11T19:34:56.678620Z","iopub.status.idle":"2024-10-11T19:34:56.841833Z","shell.execute_reply.started":"2024-10-11T19:34:56.678569Z","shell.execute_reply":"2024-10-11T19:34:56.840659Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Analysis","metadata":{}},{"cell_type":"code","source":"data.hist(figsize=(70, 40), bins=30)  \nplt.tight_layout()  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:34:56.843622Z","iopub.execute_input":"2024-10-11T19:34:56.844011Z","iopub.status.idle":"2024-10-11T19:35:19.466371Z","shell.execute_reply.started":"2024-10-11T19:34:56.843968Z","shell.execute_reply":"2024-10-11T19:35:19.465171Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## How internet usage affect physical measures ","metadata":{}},{"cell_type":"markdown","source":"### Average BMI , HeartRate , Blood pressure","metadata":{}},{"cell_type":"code","source":"averages = {\n    'BMI': data['Physical-BMI'].mean(),\n    'Heart Rate': data['Physical-HeartRate'].mean(),\n    'Systolic BP': data['Physical-Systolic_BP'].mean(),\n    'Diastolic BP': data['Physical-Diastolic_BP'].mean()\n}\n\n# Convert to DataFrame for easier plotting\naverages_df = pd.DataFrame(list(averages.items()), columns=['Feature', 'Average'])\n\n# Step 2: Create a bar plot\nplt.figure(figsize=(10, 6))\nsns.barplot(x='Feature', y='Average', data=averages_df, palette='coolwarm')\nplt.title('Average Values of BMI, Heart Rate, and Blood Pressure')\nplt.ylabel('Average')\nplt.xlabel('Features')\nplt.ylim(0, averages_df['Average'].max() + 10)  # Adjusting y-axis for better visibility\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:19.468096Z","iopub.execute_input":"2024-10-11T19:35:19.469126Z","iopub.status.idle":"2024-10-11T19:35:19.804670Z","shell.execute_reply.started":"2024-10-11T19:35:19.469047Z","shell.execute_reply":"2024-10-11T19:35:19.803704Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Childern Global Assessment Scale vs internet usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.histplot(data['CGAS-CGAS_Score'], bins=30, kde=True)  \nplt.title('Distribution of Childrens Global Assessment Scale Score')\nplt.xlabel('Assessment score')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:19.809964Z","iopub.execute_input":"2024-10-11T19:35:19.810518Z","iopub.status.idle":"2024-10-11T19:35:20.381084Z","shell.execute_reply.started":"2024-10-11T19:35:19.810472Z","shell.execute_reply":"2024-10-11T19:35:20.379817Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Assuming 'data' is your DataFrame\nmean_values = data.groupby('sii')['CGAS-CGAS_Score'].mean().reset_index()\n\n# Create a bar plot\nplt.figure(figsize=(10, 8))\nsns.barplot(x='sii', y='CGAS-CGAS_Score', data=mean_values, palette='viridis')\n\n# Customize the plot\nplt.title('Average Fitness Endurance Max Stage vs Total Internet Usage')\nplt.xlabel('Total Internet Usage (hours/day)')\nplt.ylabel('Average Fitness Endurance Max Stage')\nplt.ylim(0, 100)  # Set y-axis limits from 0 to 100\nplt.grid(axis='y')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:20.382662Z","iopub.execute_input":"2024-10-11T19:35:20.383090Z","iopub.status.idle":"2024-10-11T19:35:20.713627Z","shell.execute_reply.started":"2024-10-11T19:35:20.383044Z","shell.execute_reply":"2024-10-11T19:35:20.712478Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### BMI vs SII Severity","metadata":{}},{"cell_type":"code","source":"# Plot 1: BMI vs SII Severity\nplt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-BMI', palette='pastel')\nplt.title('BMI vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('BMI')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:20.715089Z","iopub.execute_input":"2024-10-11T19:35:20.715453Z","iopub.status.idle":"2024-10-11T19:35:21.047328Z","shell.execute_reply.started":"2024-10-11T19:35:20.715411Z","shell.execute_reply":"2024-10-11T19:35:21.046278Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Blood pressure vs internet Usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-Systolic_BP', palette='pastel')\nplt.title('Blood Pressure vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('Systolic BP')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:21.048612Z","iopub.execute_input":"2024-10-11T19:35:21.048993Z","iopub.status.idle":"2024-10-11T19:35:21.396252Z","shell.execute_reply.started":"2024-10-11T19:35:21.048952Z","shell.execute_reply":"2024-10-11T19:35:21.395106Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Heart rate vs internet usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-HeartRate', palette='pastel')\nplt.title('Heart Rate vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('Heart Rate')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:21.397699Z","iopub.execute_input":"2024-10-11T19:35:21.398103Z","iopub.status.idle":"2024-10-11T19:35:21.728252Z","shell.execute_reply.started":"2024-10-11T19:35:21.398059Z","shell.execute_reply":"2024-10-11T19:35:21.727139Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Demographics influence  ","metadata":{}},{"cell_type":"markdown","source":"### Age Distribution ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.histplot(data['Basic_Demos-Age'], bins=30, kde=True)  \nplt.title('Distribution of Age Feature')\nplt.xlabel('AGE')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:21.732901Z","iopub.execute_input":"2024-10-11T19:35:21.733316Z","iopub.status.idle":"2024-10-11T19:35:22.296684Z","shell.execute_reply.started":"2024-10-11T19:35:21.733260Z","shell.execute_reply":"2024-10-11T19:35:22.295717Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Gender Distribution","metadata":{}},{"cell_type":"code","source":"# Count the number of occurrences for each sex\nGende_counts = data['Basic_Demos-Sex'].value_counts()\n\n# Create the pie chart\nplt.figure(figsize=(7, 7))\nplt.pie(Gende_counts, labels=Gende_counts.index, autopct='%1.1f%%', startangle=90, colors=sns.color_palette('pastel'))\nplt.title('Distribution of Sex in the Dataset')\nplt.axis('equal')  # Equal aspect ratio ensures that pie chart is circular\nplt.show()\n# 0-> male  1_> female","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:22.298082Z","iopub.execute_input":"2024-10-11T19:35:22.298564Z","iopub.status.idle":"2024-10-11T19:35:22.516392Z","shell.execute_reply.started":"2024-10-11T19:35:22.298515Z","shell.execute_reply":"2024-10-11T19:35:22.515276Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_severity(score):\n    if score <= 30:\n        return 'None'\n    elif score <= 49:\n        return 'Mild'\n    elif score <= 79:\n        return 'Moderate'\n    else:\n        return 'Severe'\n\ndata['Severity'] = data['PCIAT-PCIAT_Total'].apply(categorize_severity)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:22.517942Z","iopub.execute_input":"2024-10-11T19:35:22.518410Z","iopub.status.idle":"2024-10-11T19:35:22.528208Z","shell.execute_reply.started":"2024-10-11T19:35:22.518353Z","shell.execute_reply":"2024-10-11T19:35:22.527071Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(12, 6))\nsns.boxplot(x='Severity', y='Basic_Demos-Age', data=data, palette='pastel')\nplt.title('Age Distribution Across Severity Impairment Index')\nplt.xlabel('Severity Impairment Index')\nplt.ylabel('Age')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:22.529725Z","iopub.execute_input":"2024-10-11T19:35:22.530185Z","iopub.status.idle":"2024-10-11T19:35:22.899797Z","shell.execute_reply.started":"2024-10-11T19:35:22.530129Z","shell.execute_reply":"2024-10-11T19:35:22.898469Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Calculate the average age for each severity category\naverage_age = data.groupby('Severity')['Basic_Demos-Age'].mean().reset_index()\n\n# Create a line plot\nplt.figure(figsize=(12, 6))\nsns.lineplot(x='Severity', y='Basic_Demos-Age', data=average_age, marker='o')\nplt.title('Average Age Across Severity Impairment Index')\nplt.xlabel('Severity Impairment Index')\nplt.ylabel('Average Age')\nplt.xticks(rotation=45)  # Rotate x labels for better visibility\nplt.grid(True)  # Add grid for better readability\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:22.901654Z","iopub.execute_input":"2024-10-11T19:35:22.902673Z","iopub.status.idle":"2024-10-11T19:35:23.257571Z","shell.execute_reply.started":"2024-10-11T19:35:22.902611Z","shell.execute_reply":"2024-10-11T19:35:23.256393Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ninternet_usage = data.groupby(['Basic_Demos-Age', 'PreInt_EduHx-computerinternet_hoursday']).size().unstack(fill_value=0)\n\ninternet_usage = internet_usage.reset_index()\n\ninternet_usage_melted = internet_usage.melt(id_vars='Basic_Demos-Age', \n                                              var_name='Internet Usage', \n                                              value_name='Count')\nplt.figure(figsize=(12, 6))\npalette = sns.color_palette('pastel') \nsns.barplot(data=internet_usage_melted, \n            x='Basic_Demos-Age', \n            y='Count', \n            hue='Internet Usage', \n            palette=palette)\n\nplt.xlabel('Age Group', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.title('Internet Usage Distribution by Age Group', fontsize=14)\nplt.xticks(rotation=45)\n\nhandles = []\nfor i, label in enumerate(['0=Less than 1h/day', '1=Around 1h/day', '2=Around 2hs/day', '3=More than 3hs/day']):\n    handles.append(plt.Line2D([0], [0], color=palette[i], lw=4))  # Create a line for each color\n\nplt.legend(handles, \n           ['0=Less than 1h/day', '1=Around 1h/day', '2=Around 2hs/day', '3=More than 3hs/day'], \n           title='Daily Internet Usage')\n\nplt.tight_layout() \nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:23.259178Z","iopub.execute_input":"2024-10-11T19:35:23.259578Z","iopub.status.idle":"2024-10-11T19:35:24.186377Z","shell.execute_reply.started":"2024-10-11T19:35:23.259532Z","shell.execute_reply":"2024-10-11T19:35:24.185407Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sleep Quality vs internet usage","metadata":{}},{"cell_type":"code","source":"mean_sleep_time = data.groupby('PreInt_EduHx-computerinternet_hoursday')['SDS-SDS_Total_T'].mean()\n\nplt.figure(figsize=(10, 8))\nplt.plot(mean_sleep_time.index, mean_sleep_time.values, marker='o', linestyle='-', color='blue')\n\n# Customize the plot\nplt.xlabel('Internet Usage Time (hours/day)')\nplt.ylabel('Average Sleep Time (hours)')\nplt.title('Average Sleep Time vs Internet Usage Time')\n\ncustom_ticks = [ 45, 50, 55, 60, 65, 70,75]  # Define your custom tick values\nplt.yticks(custom_ticks)\n\nplt.grid(True)\n\n# Show the plot\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:24.187691Z","iopub.execute_input":"2024-10-11T19:35:24.188094Z","iopub.status.idle":"2024-10-11T19:35:24.531663Z","shell.execute_reply.started":"2024-10-11T19:35:24.188050Z","shell.execute_reply":"2024-10-11T19:35:24.530529Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Behavioural influence ","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n\nplt.figure(figsize=(12, 6))\nsns.boxplot(x='PCIAT-PCIAT_06', y='Basic_Demos-Age', data=data, palette='pastel')\nplt.title('Age Distribution for Q: How often do your childs grades suffer because of the amount of time he or she spends online?')\nplt.xlabel('Response')\nplt.ylabel('Age')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:24.533047Z","iopub.execute_input":"2024-10-11T19:35:24.533417Z","iopub.status.idle":"2024-10-11T19:35:24.932260Z","shell.execute_reply.started":"2024-10-11T19:35:24.533374Z","shell.execute_reply":"2024-10-11T19:35:24.931199Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# List of questions to analyze\nquestions = [\n    'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', \n    'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06',\n    'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09',\n    'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12',\n    'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15',\n    'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18',\n    'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20'\n]\n\nplt.figure(figsize=(15, 40))  # Adjust the size for better visibility\n\n# Create a box plot for each question\nfor i, question in enumerate(questions):\n    plt.subplot(len(questions), 1, i + 1)  # Create subplots\n    sns.boxplot(x=question, y='Basic_Demos-Age', data=data, palette='pastel')\n    plt.title(f'Age Distribution for {question}')\n    plt.xlabel('Response')\n    plt.ylabel('Age')\n\nplt.tight_layout()  # Adjust layout to prevent overlap\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:24.933547Z","iopub.execute_input":"2024-10-11T19:35:24.933901Z","iopub.status.idle":"2024-10-11T19:35:31.329578Z","shell.execute_reply.started":"2024-10-11T19:35:24.933862Z","shell.execute_reply":"2024-10-11T19:35:31.328318Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"markdown","source":"## Handling Missing Values ","metadata":{}},{"cell_type":"markdown","source":"### Drop columns with null>50","metadata":{}},{"cell_type":"code","source":"\ncol = ['CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n       'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season',\n       'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']\ncol.extend( \n        ['Physical-Waist_Circumference','Fitness_Endurance-Max_Stage',\n         'Fitness_Endurance-Time_Mins','Fitness_Endurance-Time_Sec','FGC-FGC_GSND',\n         'FGC-FGC_GSND_Zone','FGC-FGC_GSD','FGC-FGC_GSD_Zone','BIA-BIA_Activity_Level_num',\n         'BIA-BIA_BMC','BIA-BIA_BMI','BIA-BIA_BMR','BIA-BIA_DEE','BIA-BIA_ECW','BIA-BIA_FFM',\n         'BIA-BIA_FFMI','BIA-BIA_Fat','BIA-BIA_Frame_num','BIA-BIA_ICW','BIA-BIA_LDM','BIA-BIA_LST',\n         'BIA-BIA_SMM','BIA-BIA_TBW','PAQ_A-PAQ_A_Total','PAQ_C-PAQ_C_Total'])\nlen(col)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:31.331135Z","iopub.execute_input":"2024-10-11T19:35:31.331549Z","iopub.status.idle":"2024-10-11T19:35:31.341641Z","shell.execute_reply.started":"2024-10-11T19:35:31.331502Z","shell.execute_reply":"2024-10-11T19:35:31.340432Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop(col , axis =1 ,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:31.343556Z","iopub.execute_input":"2024-10-11T19:35:31.344028Z","iopub.status.idle":"2024-10-11T19:35:31.356864Z","shell.execute_reply.started":"2024-10-11T19:35:31.343979Z","shell.execute_reply":"2024-10-11T19:35:31.355446Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:31.359096Z","iopub.execute_input":"2024-10-11T19:35:31.359636Z","iopub.status.idle":"2024-10-11T19:35:31.369317Z","shell.execute_reply.started":"2024-10-11T19:35:31.359587Z","shell.execute_reply":"2024-10-11T19:35:31.368072Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned= data.copy()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:31.371257Z","iopub.execute_input":"2024-10-11T19:35:31.371684Z","iopub.status.idle":"2024-10-11T19:35:31.379251Z","shell.execute_reply.started":"2024-10-11T19:35:31.371635Z","shell.execute_reply":"2024-10-11T19:35:31.378083Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:31.380869Z","iopub.execute_input":"2024-10-11T19:35:31.381317Z","iopub.status.idle":"2024-10-11T19:35:31.391869Z","shell.execute_reply.started":"2024-10-11T19:35:31.381257Z","shell.execute_reply":"2024-10-11T19:35:31.390715Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Filling Missing Values in first 46 columns with Knn imputer","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\ndata_cleaned['Basic_Demos-Enroll_Season'] = label_encoder.fit_transform(data_cleaned['Basic_Demos-Enroll_Season'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:31.393569Z","iopub.execute_input":"2024-10-11T19:35:31.394064Z","iopub.status.idle":"2024-10-11T19:35:31.403126Z","shell.execute_reply.started":"2024-10-11T19:35:31.394016Z","shell.execute_reply":"2024-10-11T19:35:31.401850Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\nimputer = KNNImputer(n_neighbors=5)\ndata_cleaned.iloc[:, :45] = imputer.fit_transform(data_cleaned.iloc[:, :45])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:31.412486Z","iopub.execute_input":"2024-10-11T19:35:31.412977Z","iopub.status.idle":"2024-10-11T19:35:36.990862Z","shell.execute_reply.started":"2024-10-11T19:35:31.412927Z","shell.execute_reply":"2024-10-11T19:35:36.989888Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\ndata_cleaned['Severity'] = label_encoder.fit_transform(data_cleaned['Severity'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:36.992224Z","iopub.execute_input":"2024-10-11T19:35:36.992591Z","iopub.status.idle":"2024-10-11T19:35:37.003129Z","shell.execute_reply.started":"2024-10-11T19:35:36.992547Z","shell.execute_reply":"2024-10-11T19:35:37.001959Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Filling missing values in target with Kmeans","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nimport numpy as np\nimport pandas as pd\n\ndef impute_with_kmeans(df, categorical_columns, n_clusters=4):\n    # Fill missing values temporarily with the mode (or any placeholder)\n    df_temp = df.copy()\n    for col in categorical_columns:\n        df_temp[col].fillna(df_temp[col].mode()[0], inplace=True)\n\n    # Perform KMeans clustering after filling missing values\n    kmeans = KMeans(n_clusters=n_clusters, random_state=0)\n    cluster_labels = kmeans.fit_predict(df_temp)\n\n    # Impute missing values within each cluster\n    for col in categorical_columns:\n        for cluster in np.unique(cluster_labels):\n            mask = (cluster_labels == cluster) & df[col].isna()\n            most_frequent = df.loc[cluster_labels == cluster, col].mode()[0]\n            df.loc[mask, col] = most_frequent\n\n    return df\n\n# Apply KMeans-based imputation\ndata_cleaned = impute_with_kmeans(data_cleaned, ['sii'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:37.004499Z","iopub.execute_input":"2024-10-11T19:35:37.004885Z","iopub.status.idle":"2024-10-11T19:35:37.764245Z","shell.execute_reply.started":"2024-10-11T19:35:37.004836Z","shell.execute_reply":"2024-10-11T19:35:37.762242Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned['sii'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:37.765457Z","iopub.execute_input":"2024-10-11T19:35:37.765858Z","iopub.status.idle":"2024-10-11T19:35:37.778161Z","shell.execute_reply.started":"2024-10-11T19:35:37.765807Z","shell.execute_reply":"2024-10-11T19:35:37.776828Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:37.779823Z","iopub.execute_input":"2024-10-11T19:35:37.780366Z","iopub.status.idle":"2024-10-11T19:35:37.819252Z","shell.execute_reply.started":"2024-10-11T19:35:37.780307Z","shell.execute_reply":"2024-10-11T19:35:37.818176Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:37.820884Z","iopub.execute_input":"2024-10-11T19:35:37.821255Z","iopub.status.idle":"2024-10-11T19:35:37.832133Z","shell.execute_reply.started":"2024-10-11T19:35:37.821204Z","shell.execute_reply":"2024-10-11T19:35:37.830927Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handling outliers ","metadata":{}},{"cell_type":"code","source":"outlier_indices = detect_outliers_iqr(data_cleaned)\nprint(\"Total outliers detected:\", len(set(outlier_indices)))","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:37.833627Z","iopub.execute_input":"2024-10-11T19:35:37.834108Z","iopub.status.idle":"2024-10-11T19:35:37.946877Z","shell.execute_reply.started":"2024-10-11T19:35:37.834053Z","shell.execute_reply":"2024-10-11T19:35:37.945623Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for column in data_cleaned.columns:\n    plt.figure(figsize=(8, 3))  \n    sns.boxplot(data=data_cleaned[[column]], orient='h', palette=\"Set2\")\n    \n    plt.title(f'Box Plot for {column} with Outliers', fontsize=16)\n    plt.ylabel(column, fontsize=14)\n    plt.xlabel('Values', fontsize=14)\n    \n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:37.948506Z","iopub.execute_input":"2024-10-11T19:35:37.948984Z","iopub.status.idle":"2024-10-11T19:35:54.703953Z","shell.execute_reply.started":"2024-10-11T19:35:37.948927Z","shell.execute_reply":"2024-10-11T19:35:54.702736Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate IQR for the column 'CGAS-CGAS_Score'\nQ1 = data_cleaned['CGAS-CGAS_Score'].quantile(0.25)\nQ3 = data_cleaned['CGAS-CGAS_Score'].quantile(0.75)\nIQR = Q3 - Q1\n\n# Define outlier boundaries\nlower_bound = Q1 - 1.5 * IQR\nupper_bound = Q3 + 1.5 * IQR\n\n# Filter the entire DataFrame by keeping only rows where 'CGAS-CGAS_Score' is within bounds\ndata_copy = data_cleaned[(data_cleaned['CGAS-CGAS_Score'] >= lower_bound) & (data_cleaned['CGAS-CGAS_Score'] <= upper_bound)]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:54.705805Z","iopub.execute_input":"2024-10-11T19:35:54.706831Z","iopub.status.idle":"2024-10-11T19:35:54.718627Z","shell.execute_reply.started":"2024-10-11T19:35:54.706766Z","shell.execute_reply":"2024-10-11T19:35:54.717549Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"## BMI To Heart rate Ratio","metadata":{}},{"cell_type":"code","source":"data_copy['HeartRate_BMI'] = data_copy['Physical-HeartRate'] * data_copy['Physical-BMI']","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:54.719896Z","iopub.execute_input":"2024-10-11T19:35:54.720277Z","iopub.status.idle":"2024-10-11T19:35:54.730687Z","shell.execute_reply.started":"2024-10-11T19:35:54.720236Z","shell.execute_reply":"2024-10-11T19:35:54.729517Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Max Heart rate due to age","metadata":{}},{"cell_type":"code","source":"data_copy['HRmax'] = 220 - data_copy['Basic_Demos-Age']  # Estimate HRmax based on age","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:54.732165Z","iopub.execute_input":"2024-10-11T19:35:54.732601Z","iopub.status.idle":"2024-10-11T19:35:54.745480Z","shell.execute_reply.started":"2024-10-11T19:35:54.732549Z","shell.execute_reply":"2024-10-11T19:35:54.744407Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Addation Feature ","metadata":{}},{"cell_type":"code","source":"def categorize_bmi(bmi):\n    if bmi <= 18.5:\n        return 0\n    elif bmi <= 24.9:\n        return 1\n    elif bmi <= 29.9:\n        return 2\n    elif bmi <= 40:\n        return 3\n    else:\n        return 4  \n\ndata_copy['BMI_Category'] = data_copy['Physical-BMI'].apply(categorize_bmi)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:54.747181Z","iopub.execute_input":"2024-10-11T19:35:54.747619Z","iopub.status.idle":"2024-10-11T19:35:54.759762Z","shell.execute_reply.started":"2024-10-11T19:35:54.747574Z","shell.execute_reply":"2024-10-11T19:35:54.758580Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pciat_columns = [f'PCIAT-PCIAT_{i:02d}' for i in range(1, 21)]\ndata_copy['PCIAT_Time_Management'] = data_copy[pciat_columns[:5]].mean(axis=1)\ndata_copy['PCIAT_Withdrawal_Symptoms'] = data_copy[pciat_columns[5:10]].mean(axis=1)\ndata_copy['PCIAT_Neglect_Social_Life'] = data_copy[pciat_columns[10:15]].mean(axis=1)\ndata_copy['PCIAT_Lack_Control'] = data_copy[pciat_columns[15:]].mean(axis=1)\n\n\n\ndata_copy['PCIAT_mean'] = data_copy[pciat_columns].mean(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:54.761475Z","iopub.execute_input":"2024-10-11T19:35:54.761914Z","iopub.status.idle":"2024-10-11T19:35:54.789521Z","shell.execute_reply.started":"2024-10-11T19:35:54.761867Z","shell.execute_reply":"2024-10-11T19:35:54.788277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_pciat(score):\n    if score <= 20:\n         return 0\n    elif score <= 49:\n          return 1\n    elif score <= 79:\n         return 2\n    else:\n         return 3\n    \ndata_copy['PCIAT_Category'] = data_copy['PCIAT-PCIAT_Total'].apply(categorize_pciat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:54.791184Z","iopub.execute_input":"2024-10-11T19:35:54.791682Z","iopub.status.idle":"2024-10-11T19:35:54.803403Z","shell.execute_reply.started":"2024-10-11T19:35:54.791623Z","shell.execute_reply":"2024-10-11T19:35:54.802148Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_sds(score):\n    if pd.isna(score) or score < 0:\n        return np.nan\n    elif score <= 20:\n        return 0\n    elif score <= 40:\n        return 1\n    elif score <= 60:\n        return 2\n    elif score <= 80:\n        return 3\n    else:\n        return 4  \n\ndata_copy['SDS_Severity'] = data_copy['SDS-SDS_Total_Raw'].apply(categorize_sds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:54.805026Z","iopub.execute_input":"2024-10-11T19:35:54.805529Z","iopub.status.idle":"2024-10-11T19:35:54.824648Z","shell.execute_reply.started":"2024-10-11T19:35:54.805465Z","shell.execute_reply":"2024-10-11T19:35:54.823467Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['Sleep_Quality_Index'] = (data_copy['SDS-SDS_Total_T'] - data_copy['SDS-SDS_Total_T'].min()) / (data_copy['SDS-SDS_Total_T'].max() - data_copy['SDS-SDS_Total_T'].min())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:54.826371Z","iopub.execute_input":"2024-10-11T19:35:54.827191Z","iopub.status.idle":"2024-10-11T19:35:54.837563Z","shell.execute_reply.started":"2024-10-11T19:35:54.827131Z","shell.execute_reply":"2024-10-11T19:35:54.836487Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['Physical_Health_Index'] = ((data_copy['Physical-BMI'] - data_copy['Physical-BMI'].mean()) / data_copy['Physical-BMI'].std() + (data_copy['Physical-Systolic_BP'] - data_copy['Physical-Systolic_BP'].mean()) / data_copy['Physical-Systolic_BP'].std() + (data_copy['Physical-HeartRate'] - data_copy['Physical-HeartRate'].mean()) / data_copy['Physical-HeartRate'].std()) / 3\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:54.839034Z","iopub.execute_input":"2024-10-11T19:35:54.839381Z","iopub.status.idle":"2024-10-11T19:35:54.850310Z","shell.execute_reply.started":"2024-10-11T19:35:54.839342Z","shell.execute_reply":"2024-10-11T19:35:54.849278Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['Sleep_Quality_Index'] = (data_copy['SDS-SDS_Total_T'] - data_copy['SDS-SDS_Total_T'].min()) / (data_copy['SDS-SDS_Total_T'].max() - data_copy['SDS-SDS_Total_T'].min())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:54.851906Z","iopub.execute_input":"2024-10-11T19:35:54.852378Z","iopub.status.idle":"2024-10-11T19:35:54.862669Z","shell.execute_reply.started":"2024-10-11T19:35:54.852322Z","shell.execute_reply":"2024-10-11T19:35:54.861517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fgc_columns = ['FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_TL']\n\ndata_copy['Overall_Fitness_Score'] = data_copy[fgc_columns].mean(axis=1)\ndata_copy['Internet_Usage_Score'] = data_copy['PCIAT-PCIAT_Total'] / 100\ndata_copy['Physical_Activity_Score'] = data_copy['Overall_Fitness_Score'] / data_copy['Overall_Fitness_Score'].max()\ndata_copy['Lifestyle_Score'] = ((1 - data_copy['Internet_Usage_Score']) +  data_copy['Physical_Activity_Score'] + (1 - data_copy['Sleep_Quality_Index'])) / 3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:54.864212Z","iopub.execute_input":"2024-10-11T19:35:54.864618Z","iopub.status.idle":"2024-10-11T19:35:54.882361Z","shell.execute_reply.started":"2024-10-11T19:35:54.864574Z","shell.execute_reply":"2024-10-11T19:35:54.881214Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_bp(systolic, diastolic):\n    if systolic < 120 and diastolic < 80:\n        return 0\n    elif 120 <= systolic < 130 and diastolic < 80:\n        return 1\n    elif 130 <= systolic < 140 or 80 <= diastolic < 90:\n        return 2\n    else:\n        return 3 \n\ndata_copy['BP_Category'] = data_copy.apply(lambda row: categorize_bp(row['Physical-Systolic_BP'], row['Physical-Diastolic_BP']), axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:54.884165Z","iopub.execute_input":"2024-10-11T19:35:54.884637Z","iopub.status.idle":"2024-10-11T19:35:54.964330Z","shell.execute_reply.started":"2024-10-11T19:35:54.884578Z","shell.execute_reply":"2024-10-11T19:35:54.963101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['BMI_Age_Interaction'] = data_copy['Physical-BMI'] * data_copy['Basic_Demos-Age']\ndata_copy['HeartRate_BPCategory_Interaction'] = data_copy['BP_Category'] * data_copy['Basic_Demos-Age']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:54.965815Z","iopub.execute_input":"2024-10-11T19:35:54.966206Z","iopub.status.idle":"2024-10-11T19:35:54.978588Z","shell.execute_reply.started":"2024-10-11T19:35:54.966163Z","shell.execute_reply":"2024-10-11T19:35:54.977358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy = data_copy.drop(columns=['SDS-SDS_Total_T', 'SDS-SDS_Total_Raw', 'PCIAT-PCIAT_Total', 'Physical-Height', 'Physical-Weight', 'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_TL', 'FGC-FGC_CU_Zone', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_TL_Zone', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20'])\n\n\n\ndata_copy = data_copy.drop(columns=['PCIAT_Time_Management', 'PCIAT_Withdrawal_Symptoms', 'PCIAT_Neglect_Social_Life', 'PCIAT_Lack_Control',  'Basic_Demos-Age', 'Physical-BMI'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:54.980181Z","iopub.execute_input":"2024-10-11T19:35:54.980628Z","iopub.status.idle":"2024-10-11T19:35:54.993531Z","shell.execute_reply.started":"2024-10-11T19:35:54.980580Z","shell.execute_reply":"2024-10-11T19:35:54.992334Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Body Strength & Flexibility","metadata":{}},{"cell_type":"code","source":"# data_copy['BodyStrength&Flexibility'] = data_copy['FGC-FGC_CU'] + data_copy['FGC-FGC_PU'] + data_copy['FGC-FGC_SRL'] + data_copy['FGC-FGC_SRR'] + data_copy['FGC-FGC_TL']\n# data_copy['BodyStrength&Flexibility_class'] = data_copy['FGC-FGC_CU_Zone'] + data_copy['FGC-FGC_PU_Zone'] + data_copy['FGC-FGC_SRL_Zone'] + data_copy['FGC-FGC_SRR_Zone'] + data_copy['FGC-FGC_TL_Zone']\n\n# data_copy = data_copy.drop(['FGC-FGC_CU','FGC-FGC_PU','FGC-FGC_SRL','FGC-FGC_SRR','FGC-FGC_TL'], axis=1)\n\n# data_copy = data_copy.drop(['FGC-FGC_CU_Zone','FGC-FGC_PU_Zone','FGC-FGC_SRL_Zone','FGC-FGC_SRR_Zone','FGC-FGC_TL_Zone'], axis=1)                          \n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:54.995161Z","iopub.execute_input":"2024-10-11T19:35:54.995586Z","iopub.status.idle":"2024-10-11T19:35:55.002687Z","shell.execute_reply.started":"2024-10-11T19:35:54.995539Z","shell.execute_reply":"2024-10-11T19:35:55.001562Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:55.004043Z","iopub.execute_input":"2024-10-11T19:35:55.004438Z","iopub.status.idle":"2024-10-11T19:35:55.018622Z","shell.execute_reply.started":"2024-10-11T19:35:55.004381Z","shell.execute_reply":"2024-10-11T19:35:55.017263Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Transfomation and scaling","metadata":{}},{"cell_type":"markdown","source":"## Data Splitting ","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = data_copy.drop('sii', axis=1)  \ny = data_copy['sii']  \nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)\nprint(X_train.shape, X_test.shape, y_train.shape, y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:55.019983Z","iopub.execute_input":"2024-10-11T19:35:55.020374Z","iopub.status.idle":"2024-10-11T19:35:55.035532Z","shell.execute_reply.started":"2024-10-11T19:35:55.020331Z","shell.execute_reply":"2024-10-11T19:35:55.034122Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"value_counts = data_copy['sii'].value_counts()\n\n# Plotting the value counts\nplt.figure(figsize=(8, 6))\nvalue_counts.plot(kind='bar', color='skyblue')\nplt.title(\"Value Counts of 'sii'\")\nplt.xlabel('Categories')\nplt.ylabel('Counts')\nplt.xticks(rotation=0)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:55.036808Z","iopub.execute_input":"2024-10-11T19:35:55.037162Z","iopub.status.idle":"2024-10-11T19:35:55.344145Z","shell.execute_reply.started":"2024-10-11T19:35:55.037122Z","shell.execute_reply":"2024-10-11T19:35:55.342948Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handling imbalanced data ","metadata":{}},{"cell_type":"code","source":"from collections import Counter\nfrom imblearn.over_sampling import SMOTE\n\n# Check the class distribution before applying SMOTE\nclass_distribution_before = Counter(y_train)\nprint(\"Class distribution before SMOTE:\", class_distribution_before)\n\n# Initialize SMOTE with specified parameters\nsmote = SMOTE(sampling_strategy='auto', random_state=42)\n\n# Apply SMOTE to the training data\nX_train_resampled, y_train_resampled = smote.fit_resample(X_train, y_train)\n\n# Check the class distribution after applying SMOTE\nclass_distribution_after = Counter(y_train_resampled)\nprint(\"Class distribution after SMOTE:\", class_distribution_after)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:55.345415Z","iopub.execute_input":"2024-10-11T19:35:55.345770Z","iopub.status.idle":"2024-10-11T19:35:55.387270Z","shell.execute_reply.started":"2024-10-11T19:35:55.345717Z","shell.execute_reply":"2024-10-11T19:35:55.386190Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train = y_train_resampled","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:55.388573Z","iopub.execute_input":"2024-10-11T19:35:55.388969Z","iopub.status.idle":"2024-10-11T19:35:55.394592Z","shell.execute_reply.started":"2024-10-11T19:35:55.388926Z","shell.execute_reply":"2024-10-11T19:35:55.393454Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Scaling","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import RobustScaler\nscaler = RobustScaler()  # or StandardScaler()\n\n# Fit the scaler on X_train and transform both X_train and X_test\nX_train = pd.DataFrame(scaler.fit_transform(X_train_resampled), columns=X_train_resampled.columns)\nX_test = pd.DataFrame(scaler.transform(X_test), columns=X_test.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:55.396053Z","iopub.execute_input":"2024-10-11T19:35:55.396441Z","iopub.status.idle":"2024-10-11T19:35:55.427724Z","shell.execute_reply.started":"2024-10-11T19:35:55.396398Z","shell.execute_reply":"2024-10-11T19:35:55.426606Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.shape\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:55.429412Z","iopub.execute_input":"2024-10-11T19:35:55.430265Z","iopub.status.idle":"2024-10-11T19:35:55.438087Z","shell.execute_reply.started":"2024-10-11T19:35:55.430191Z","shell.execute_reply":"2024-10-11T19:35:55.436840Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:35:55.439984Z","iopub.execute_input":"2024-10-11T19:35:55.440585Z","iopub.status.idle":"2024-10-11T19:35:55.451083Z","shell.execute_reply.started":"2024-10-11T19:35:55.440526Z","shell.execute_reply":"2024-10-11T19:35:55.449934Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PCA","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA \npca = PCA(n_components= 20 )  # Reduce to 2 components\nX_train = pca.fit_transform(X_train)\nX_test = pca.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:55.452726Z","iopub.execute_input":"2024-10-11T19:35:55.453261Z","iopub.status.idle":"2024-10-11T19:35:56.327922Z","shell.execute_reply.started":"2024-10-11T19:35:55.453190Z","shell.execute_reply":"2024-10-11T19:35:56.325671Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modeling ","metadata":{}},{"cell_type":"code","source":"pip install lazypredict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:35:56.330237Z","iopub.execute_input":"2024-10-11T19:35:56.330599Z","iopub.status.idle":"2024-10-11T19:36:10.829991Z","shell.execute_reply.started":"2024-10-11T19:35:56.330557Z","shell.execute_reply":"2024-10-11T19:36:10.828524Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lazypredict.Supervised import LazyClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.datasets import load_iris\nimport pandas as pd\n\n\n\nclf = LazyClassifier(verbose=0, ignore_warnings=True, custom_metric=None)\n\nmodels, predictions = clf.fit(X_train, X_test, y_train, y_test)\n\nprint(models)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:36:10.832058Z","iopub.execute_input":"2024-10-11T19:36:10.832587Z","iopub.status.idle":"2024-10-11T19:36:43.124828Z","shell.execute_reply.started":"2024-10-11T19:36:10.832515Z","shell.execute_reply":"2024-10-11T19:36:43.123664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result1, result2, result3 = [], [], [] ","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:36:43.126608Z","iopub.execute_input":"2024-10-11T19:36:43.127012Z","iopub.status.idle":"2024-10-11T19:36:43.132151Z","shell.execute_reply.started":"2024-10-11T19:36:43.126966Z","shell.execute_reply":"2024-10-11T19:36:43.130916Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_f1_scores = []\ntest_f1_scores = []\nmodel_names = []\n\ndef modeling(model, model_name):\n    # Fit the model using the scaled resampled training data\n    global train_f1_scores, test_f1_scores, model_names\n\n    model.fit(X_train, y_train_resampled)  # Use y_train_resampled here\n    \n    # Predictions\n    train_pred = model.predict(X_train)    # Predictions on the scaled resampled train set\n    test_pred = model.predict(X_test)      # Predictions on the scaled original test set\n    \n    # Calculate metrics with specified average\n    train_accuracy = accuracy_score(y_train_resampled, train_pred) * 100\n    train_recall = recall_score(y_train_resampled, train_pred, average='weighted') * 100\n    train_f1_score = f1_score(y_train_resampled, train_pred, average='weighted') * 100\n    \n    test_accuracy = accuracy_score(y_test, test_pred) * 100\n    test_recall = recall_score(y_test, test_pred, average='weighted') * 100\n    test_f1_score = f1_score(y_test, test_pred, average='weighted') * 100\n\n\n    train_f1_scores.append(train_f1_score)\n    test_f1_scores.append(test_f1_score)\n    model_names.append(model_name)\n\n    \n    # Append results\n    result1.append(test_accuracy)\n    result2.append(test_recall)\n    result3.append(test_f1_score)\n    \n    print(\"Classification Report for Test Data:\")\n    print(classification_report(y_test, test_pred))\n    \n    print(\"\\nClassification Report for Scaled Resampled Train Data:\")\n    print(classification_report(y_train_resampled, train_pred))  # Use y_train_resampled here\n    \n    # Accuracy, Recall, and F1 Scores\n    print(f'Training Accuracy: {train_accuracy}, Train Recall: {train_recall}, Train F1: {train_f1_score}')\n    print(f'Test Accuracy: {test_accuracy}, Test Recall: {test_recall}, Test F1: {test_f1_score}')\n    \n    # Confusion matrix\n    cm = confusion_matrix(y_test, test_pred)\n    sns.heatmap(cm, annot=True, fmt='0.2f', cmap='YlGnBu', linewidths=1)\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n    plt.title('Confusion Matrix')\n    plt.show()\n\n    \n\n    labels = ['Train F1 Score', 'Test F1 Score']\n    scores = [train_f1_score, test_f1_score]\n\n    plt.figure(figsize=(6, 4))\n    plt.bar(labels, scores, color=['#1f77b4', '#ff7f0e'])\n    plt.ylim(0, 100)  # Since F1 scores are percentages, 0 to 100 is the range\n    plt.title(f'F1 Score Comparison: {model_name}')\n    plt.ylabel('F1 Score (%)')\n\n    # Display values on top of bars\n    for i, score in enumerate(scores):\n        plt.text(i, score + 1, f'{score:.2f}%', ha='center')\n\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:36:43.133952Z","iopub.execute_input":"2024-10-11T19:36:43.134356Z","iopub.status.idle":"2024-10-11T19:36:43.151922Z","shell.execute_reply.started":"2024-10-11T19:36:43.134311Z","shell.execute_reply":"2024-10-11T19:36:43.150632Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression  \nlogistic_regression = LogisticRegression()\nmodeling(logistic_regression, 'Logistic Regression')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:36:43.153830Z","iopub.execute_input":"2024-10-11T19:36:43.154933Z","iopub.status.idle":"2024-10-11T19:36:44.580431Z","shell.execute_reply.started":"2024-10-11T19:36:43.154866Z","shell.execute_reply":"2024-10-11T19:36:44.579506Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SVM = SVC()\nmodeling(SVM, 'SVM')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:36:44.581824Z","iopub.execute_input":"2024-10-11T19:36:44.582201Z","iopub.status.idle":"2024-10-11T19:36:46.122134Z","shell.execute_reply.started":"2024-10-11T19:36:44.582161Z","shell.execute_reply":"2024-10-11T19:36:46.121133Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decision_tree_classifier = DecisionTreeClassifier()\nmodeling(decision_tree_classifier, 'Decision Tree')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:36:46.123989Z","iopub.execute_input":"2024-10-11T19:36:46.124358Z","iopub.status.idle":"2024-10-11T19:36:47.008000Z","shell.execute_reply.started":"2024-10-11T19:36:46.124318Z","shell.execute_reply":"2024-10-11T19:36:47.006767Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decision_tree_classifier = DecisionTreeClassifier( max_depth= 7 , min_samples_split= 7)\nmodeling(decision_tree_classifier, 'Decision Tree Tuned')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:36:47.009733Z","iopub.execute_input":"2024-10-11T19:36:47.010697Z","iopub.status.idle":"2024-10-11T19:36:47.831215Z","shell.execute_reply.started":"2024-10-11T19:36:47.010632Z","shell.execute_reply":"2024-10-11T19:36:47.830004Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sgd_classifier = SGDClassifier(max_iter = 500) \nmodeling(sgd_classifier, 'SGD')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:36:47.833190Z","iopub.execute_input":"2024-10-11T19:36:47.833749Z","iopub.status.idle":"2024-10-11T19:36:48.741944Z","shell.execute_reply.started":"2024-10-11T19:36:47.833687Z","shell.execute_reply":"2024-10-11T19:36:48.740628Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nknn = KNeighborsClassifier()\nmodeling(knn, 'KNN Classifier')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:36:48.743439Z","iopub.execute_input":"2024-10-11T19:36:48.743852Z","iopub.status.idle":"2024-10-11T19:36:50.198052Z","shell.execute_reply.started":"2024-10-11T19:36:48.743804Z","shell.execute_reply":"2024-10-11T19:36:50.196888Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n\n# Create an instance of GaussianNB\nnaive_bayes = GaussianNB()\n\n# Call your modeling function\nmodeling(naive_bayes, 'Naive Bayes')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:36:50.199586Z","iopub.execute_input":"2024-10-11T19:36:50.199981Z","iopub.status.idle":"2024-10-11T19:36:50.912318Z","shell.execute_reply.started":"2024-10-11T19:36:50.199937Z","shell.execute_reply":"2024-10-11T19:36:50.911121Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nrandom_forest = RandomForestClassifier(random_state=1)\nmodeling(random_forest, 'Random Forest')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:36:50.913932Z","iopub.execute_input":"2024-10-11T19:36:50.914277Z","iopub.status.idle":"2024-10-11T19:36:54.452445Z","shell.execute_reply.started":"2024-10-11T19:36:50.914238Z","shell.execute_reply":"2024-10-11T19:36:54.451291Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"adaBosster=AdaBoostClassifier( n_estimators=50)\nmodeling(adaBosster, 'AdaBoost')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:36:54.454274Z","iopub.execute_input":"2024-10-11T19:36:54.455188Z","iopub.status.idle":"2024-10-11T19:36:56.443632Z","shell.execute_reply.started":"2024-10-11T19:36:54.455123Z","shell.execute_reply":"2024-10-11T19:36:56.442501Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingClassifier\n\ngradient=GradientBoostingClassifier()\nmodeling(gradient, 'Gradient Boost')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T19:36:56.455966Z","iopub.execute_input":"2024-10-11T19:36:56.456400Z","iopub.status.idle":"2024-10-11T19:37:18.180916Z","shell.execute_reply.started":"2024-10-11T19:36:56.456358Z","shell.execute_reply":"2024-10-11T19:37:18.179940Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_f1_scores():\n    plt.figure(figsize=(12, 6))\n    x = range(len(model_names))\n    \n    plt.plot(x, train_f1_scores, 'bo-', label='Training F1 Score')\n    plt.plot(x, test_f1_scores, 'ro-', label='Testing F1 Score')\n    \n    plt.xlabel('Models')\n    plt.ylabel('F1 Score (Weighted)')\n    plt.title('Training and Testing F1 Scores for Different Models')\n    plt.xticks(x, model_names, rotation=45, ha='right')\n    plt.legend()\n    plt.tight_layout()\n    plt.show()\n\ndef plot_f1_scores1():\n    plt.figure(figsize=(12, 6))\n    \n    x = np.arange(len(model_names))  # the label locations\n    width = 0.35  # the width of the bars\n    \n    # Create the bars\n    rects1 = plt.bar(x - width/2, train_f1_scores, width, label='Train', color='blue', alpha=0.7)\n    rects2 = plt.bar(x + width/2, test_f1_scores, width, label='Test', color='red', alpha=0.7)\n\n    # Add some text for labels, title and custom x-axis tick labels, etc.\n    plt.ylabel('F1 Score (Weighted)')\n    plt.title('F1 Scores for Different Models (Training and Testing)')\n    plt.xticks(x, model_names, rotation=45, ha='right')\n    plt.legend()\n\n    # Add value labels on the bars\n    def autolabel(rects):\n        for rect in rects:\n            height = rect.get_height()\n            plt.annotate(f'{height:.1f}',\n                        xy=(rect.get_x() + rect.get_width() / 2, height),\n                        xytext=(0, 3),  # 3 points vertical offset\n                        textcoords=\"offset points\",\n                        ha='center', va='bottom')\n\n    autolabel(rects1)\n    autolabel(rects2)\n\n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:37:18.182197Z","iopub.execute_input":"2024-10-11T19:37:18.182574Z","iopub.status.idle":"2024-10-11T19:37:18.197124Z","shell.execute_reply.started":"2024-10-11T19:37:18.182531Z","shell.execute_reply":"2024-10-11T19:37:18.195957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_f1_scores()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:37:18.198657Z","iopub.execute_input":"2024-10-11T19:37:18.199081Z","iopub.status.idle":"2024-10-11T19:37:18.718917Z","shell.execute_reply.started":"2024-10-11T19:37:18.199037Z","shell.execute_reply":"2024-10-11T19:37:18.717730Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_f1_scores1()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T19:37:18.720520Z","iopub.execute_input":"2024-10-11T19:37:18.720969Z","iopub.status.idle":"2024-10-11T19:37:19.373871Z","shell.execute_reply.started":"2024-10-11T19:37:18.720923Z","shell.execute_reply":"2024-10-11T19:37:19.372904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}