{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt \nimport pandas as pd \nimport plotly.express as px\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport math\nfrom plotly.subplots import make_subplots\nimport numpy as np\nfrom numpy import linalg as LA\nimport plotly.express as px\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\nimport warnings\n\nwarnings.filterwarnings('ignore')\n\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nfrom sklearn.svm import SVC\nimport numpy as np\nimport matplotlib.pyplot as plt \nimport pandas as pd \nimport plotly.express as px\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport math\nfrom plotly.subplots import make_subplots\nimport numpy as np\nfrom numpy import linalg as LA\nimport plotly.express as px\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\n\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import recall_score, f1_score, accuracy_score\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import (accuracy_score, recall_score, f1_score, \n                             classification_report, confusion_matrix)\nfrom sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:16.463089Z","iopub.execute_input":"2024-10-11T16:47:16.463826Z","iopub.status.idle":"2024-10-11T16:47:16.473824Z","shell.execute_reply.started":"2024-10-11T16:47:16.463785Z","shell.execute_reply":"2024-10-11T16:47:16.472636Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Exploration ","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:16.475699Z","iopub.execute_input":"2024-10-11T16:47:16.476042Z","iopub.status.idle":"2024-10-11T16:47:16.528685Z","shell.execute_reply.started":"2024-10-11T16:47:16.476007Z","shell.execute_reply":"2024-10-11T16:47:16.527674Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:16.529970Z","iopub.execute_input":"2024-10-11T16:47:16.530285Z","iopub.status.idle":"2024-10-11T16:47:16.536874Z","shell.execute_reply.started":"2024-10-11T16:47:16.530252Z","shell.execute_reply":"2024-10-11T16:47:16.535666Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:16.539782Z","iopub.execute_input":"2024-10-11T16:47:16.540509Z","iopub.status.idle":"2024-10-11T16:47:16.586637Z","shell.execute_reply.started":"2024-10-11T16:47:16.540460Z","shell.execute_reply":"2024-10-11T16:47:16.585432Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:16.587914Z","iopub.execute_input":"2024-10-11T16:47:16.588252Z","iopub.status.idle":"2024-10-11T16:47:16.606847Z","shell.execute_reply.started":"2024-10-11T16:47:16.588217Z","shell.execute_reply":"2024-10-11T16:47:16.605659Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop('id', axis=1 , inplace = True)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:16.608181Z","iopub.execute_input":"2024-10-11T16:47:16.608494Z","iopub.status.idle":"2024-10-11T16:47:16.621032Z","shell.execute_reply.started":"2024-10-11T16:47:16.608452Z","shell.execute_reply":"2024-10-11T16:47:16.619960Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:16.622216Z","iopub.execute_input":"2024-10-11T16:47:16.622543Z","iopub.status.idle":"2024-10-11T16:47:16.653433Z","shell.execute_reply.started":"2024-10-11T16:47:16.622509Z","shell.execute_reply":"2024-10-11T16:47:16.652434Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop_duplicates(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:16.654742Z","iopub.execute_input":"2024-10-11T16:47:16.655137Z","iopub.status.idle":"2024-10-11T16:47:16.679916Z","shell.execute_reply.started":"2024-10-11T16:47:16.655102Z","shell.execute_reply":"2024-10-11T16:47:16.679000Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:16.681015Z","iopub.execute_input":"2024-10-11T16:47:16.681310Z","iopub.status.idle":"2024-10-11T16:47:16.709560Z","shell.execute_reply.started":"2024-10-11T16:47:16.681278Z","shell.execute_reply":"2024-10-11T16:47:16.708359Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:16.712924Z","iopub.execute_input":"2024-10-11T16:47:16.713265Z","iopub.status.idle":"2024-10-11T16:47:17.105797Z","shell.execute_reply.started":"2024-10-11T16:47:16.713229Z","shell.execute_reply":"2024-10-11T16:47:17.104775Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:17.107191Z","iopub.execute_input":"2024-10-11T16:47:17.107639Z","iopub.status.idle":"2024-10-11T16:47:17.119755Z","shell.execute_reply.started":"2024-10-11T16:47:17.107572Z","shell.execute_reply":"2024-10-11T16:47:17.118667Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols = [col for col in data.columns if data[col].dtype != 'O']\ncat_cols = [col for col in data.columns if col not in num_cols]\nprint(f'Numerical columns: {num_cols}')\nprint(\"-------------------\")\nprint(f'Categorical columns: {cat_cols}')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:17.121439Z","iopub.execute_input":"2024-10-11T16:47:17.122075Z","iopub.status.idle":"2024-10-11T16:47:17.129087Z","shell.execute_reply.started":"2024-10-11T16:47:17.122024Z","shell.execute_reply":"2024-10-11T16:47:17.128026Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def detect_outliers_iqr(df):\n    outlier_indices = []\n\n    for column in df.select_dtypes(include=['float64', 'int64']).columns:\n        Q1 = df[column].quantile(0.25)\n        Q3 = df[column].quantile(0.75)\n        IQR = Q3 - Q1\n\n        lower_bound = Q1 - 1.5 * IQR\n        upper_bound = Q3 + 1.5 * IQR\n\n        outliers = df[(df[column] < lower_bound) | (df[column] > upper_bound)]\n        outlier_indices.extend(outliers.index)\n\n        print(f'Outliers in {column}:', outliers.shape[0])\n\n    return outlier_indices","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:17.130097Z","iopub.execute_input":"2024-10-11T16:47:17.130389Z","iopub.status.idle":"2024-10-11T16:47:17.143205Z","shell.execute_reply.started":"2024-10-11T16:47:17.130357Z","shell.execute_reply":"2024-10-11T16:47:17.142224Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"outlier_indices = detect_outliers_iqr(data)\nprint(\"Total outliers detected:\", len(set(outlier_indices)))","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:17.146230Z","iopub.execute_input":"2024-10-11T16:47:17.146541Z","iopub.status.idle":"2024-10-11T16:47:17.291688Z","shell.execute_reply.started":"2024-10-11T16:47:17.146507Z","shell.execute_reply":"2024-10-11T16:47:17.290672Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Analysis","metadata":{}},{"cell_type":"code","source":"data.hist(figsize=(70, 40), bins=30)  \nplt.tight_layout()  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:17.294933Z","iopub.execute_input":"2024-10-11T16:47:17.295279Z","iopub.status.idle":"2024-10-11T16:47:36.381501Z","shell.execute_reply.started":"2024-10-11T16:47:17.295243Z","shell.execute_reply":"2024-10-11T16:47:36.380340Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## How internet usage affect physical measures ","metadata":{}},{"cell_type":"markdown","source":"### Average BMI , HeartRate , Blood pressure","metadata":{}},{"cell_type":"code","source":"averages = {\n    'BMI': data['Physical-BMI'].mean(),\n    'Heart Rate': data['Physical-HeartRate'].mean(),\n    'Systolic BP': data['Physical-Systolic_BP'].mean(),\n    'Diastolic BP': data['Physical-Diastolic_BP'].mean()\n}\n\n# Convert to DataFrame for easier plotting\naverages_df = pd.DataFrame(list(averages.items()), columns=['Feature', 'Average'])\n\n# Step 2: Create a bar plot\nplt.figure(figsize=(10, 6))\nsns.barplot(x='Feature', y='Average', data=averages_df, palette='coolwarm')\nplt.title('Average Values of BMI, Heart Rate, and Blood Pressure')\nplt.ylabel('Average')\nplt.xlabel('Features')\nplt.ylim(0, averages_df['Average'].max() + 10)  # Adjusting y-axis for better visibility\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:36.383096Z","iopub.execute_input":"2024-10-11T16:47:36.383480Z","iopub.status.idle":"2024-10-11T16:47:36.677017Z","shell.execute_reply.started":"2024-10-11T16:47:36.383439Z","shell.execute_reply":"2024-10-11T16:47:36.675975Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Childern Global Assessment Scale vs internet usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.histplot(data['CGAS-CGAS_Score'], bins=30, kde=True)  \nplt.title('Distribution of Childrens Global Assessment Scale Score')\nplt.xlabel('Assessment score')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:36.678443Z","iopub.execute_input":"2024-10-11T16:47:36.678837Z","iopub.status.idle":"2024-10-11T16:47:37.211749Z","shell.execute_reply.started":"2024-10-11T16:47:36.678803Z","shell.execute_reply":"2024-10-11T16:47:37.210285Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Assuming 'data' is your DataFrame\nmean_values = data.groupby('sii')['CGAS-CGAS_Score'].mean().reset_index()\n\n# Create a bar plot\nplt.figure(figsize=(10, 8))\nsns.barplot(x='sii', y='CGAS-CGAS_Score', data=mean_values, palette='viridis')\n\n# Customize the plot\nplt.title('Average Fitness Endurance Max Stage vs Total Internet Usage')\nplt.xlabel('Total Internet Usage (hours/day)')\nplt.ylabel('Average Fitness Endurance Max Stage')\nplt.ylim(0, 100)  # Set y-axis limits from 0 to 100\nplt.grid(axis='y')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:37.213056Z","iopub.execute_input":"2024-10-11T16:47:37.213374Z","iopub.status.idle":"2024-10-11T16:47:37.514979Z","shell.execute_reply.started":"2024-10-11T16:47:37.213341Z","shell.execute_reply":"2024-10-11T16:47:37.513859Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### BMI vs SII Severity","metadata":{}},{"cell_type":"code","source":"# Plot 1: BMI vs SII Severity\nplt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-BMI', palette='pastel')\nplt.title('BMI vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('BMI')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:37.516552Z","iopub.execute_input":"2024-10-11T16:47:37.517018Z","iopub.status.idle":"2024-10-11T16:47:37.814410Z","shell.execute_reply.started":"2024-10-11T16:47:37.516969Z","shell.execute_reply":"2024-10-11T16:47:37.813510Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Blood pressure vs internet Usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-Systolic_BP', palette='pastel')\nplt.title('Blood Pressure vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('Systolic BP')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:37.816089Z","iopub.execute_input":"2024-10-11T16:47:37.816407Z","iopub.status.idle":"2024-10-11T16:47:38.134542Z","shell.execute_reply.started":"2024-10-11T16:47:37.816374Z","shell.execute_reply":"2024-10-11T16:47:38.133646Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Heart rate vs internet usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-HeartRate', palette='pastel')\nplt.title('Heart Rate vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('Heart Rate')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:38.135761Z","iopub.execute_input":"2024-10-11T16:47:38.136076Z","iopub.status.idle":"2024-10-11T16:47:38.433880Z","shell.execute_reply.started":"2024-10-11T16:47:38.136043Z","shell.execute_reply":"2024-10-11T16:47:38.432868Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Demographics influence  ","metadata":{}},{"cell_type":"markdown","source":"### Age Distribution ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.histplot(data['Basic_Demos-Age'], bins=30, kde=True)  \nplt.title('Distribution of Age Feature')\nplt.xlabel('AGE')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:38.435094Z","iopub.execute_input":"2024-10-11T16:47:38.435393Z","iopub.status.idle":"2024-10-11T16:47:38.965793Z","shell.execute_reply.started":"2024-10-11T16:47:38.435362Z","shell.execute_reply":"2024-10-11T16:47:38.964580Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Gender Distribution","metadata":{}},{"cell_type":"code","source":"# Count the number of occurrences for each sex\nGende_counts = data['Basic_Demos-Sex'].value_counts()\n\n# Create the pie chart\nplt.figure(figsize=(7, 7))\nplt.pie(Gende_counts, labels=Gende_counts.index, autopct='%1.1f%%', startangle=90, colors=sns.color_palette('pastel'))\nplt.title('Distribution of Sex in the Dataset')\nplt.axis('equal')  # Equal aspect ratio ensures that pie chart is circular\nplt.show()\n# 0-> male  1_> female","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:38.967020Z","iopub.execute_input":"2024-10-11T16:47:38.967349Z","iopub.status.idle":"2024-10-11T16:47:39.173660Z","shell.execute_reply.started":"2024-10-11T16:47:38.967314Z","shell.execute_reply":"2024-10-11T16:47:39.172638Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_severity(score):\n    if score <= 30:\n        return 'None'\n    elif score <= 49:\n        return 'Mild'\n    elif score <= 79:\n        return 'Moderate'\n    else:\n        return 'Severe'\n\ndata['Severity'] = data['PCIAT-PCIAT_Total'].apply(categorize_severity)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:39.175246Z","iopub.execute_input":"2024-10-11T16:47:39.175696Z","iopub.status.idle":"2024-10-11T16:47:39.186175Z","shell.execute_reply.started":"2024-10-11T16:47:39.175649Z","shell.execute_reply":"2024-10-11T16:47:39.185097Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(12, 6))\nsns.boxplot(x='Severity', y='Basic_Demos-Age', data=data, palette='pastel')\nplt.title('Age Distribution Across Severity Impairment Index')\nplt.xlabel('Severity Impairment Index')\nplt.ylabel('Age')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:39.187359Z","iopub.execute_input":"2024-10-11T16:47:39.187713Z","iopub.status.idle":"2024-10-11T16:47:39.520554Z","shell.execute_reply.started":"2024-10-11T16:47:39.187671Z","shell.execute_reply":"2024-10-11T16:47:39.519481Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Calculate the average age for each severity category\naverage_age = data.groupby('Severity')['Basic_Demos-Age'].mean().reset_index()\n\n# Create a line plot\nplt.figure(figsize=(12, 6))\nsns.lineplot(x='Severity', y='Basic_Demos-Age', data=average_age, marker='o')\nplt.title('Average Age Across Severity Impairment Index')\nplt.xlabel('Severity Impairment Index')\nplt.ylabel('Average Age')\nplt.xticks(rotation=45)  # Rotate x labels for better visibility\nplt.grid(True)  # Add grid for better readability\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:39.527748Z","iopub.execute_input":"2024-10-11T16:47:39.528113Z","iopub.status.idle":"2024-10-11T16:47:39.856519Z","shell.execute_reply.started":"2024-10-11T16:47:39.528079Z","shell.execute_reply":"2024-10-11T16:47:39.855216Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ninternet_usage = data.groupby(['Basic_Demos-Age', 'PreInt_EduHx-computerinternet_hoursday']).size().unstack(fill_value=0)\n\ninternet_usage = internet_usage.reset_index()\n\ninternet_usage_melted = internet_usage.melt(id_vars='Basic_Demos-Age', \n                                              var_name='Internet Usage', \n                                              value_name='Count')\nplt.figure(figsize=(12, 6))\npalette = sns.color_palette('pastel') \nsns.barplot(data=internet_usage_melted, \n            x='Basic_Demos-Age', \n            y='Count', \n            hue='Internet Usage', \n            palette=palette)\n\nplt.xlabel('Age Group', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.title('Internet Usage Distribution by Age Group', fontsize=14)\nplt.xticks(rotation=45)\n\nhandles = []\nfor i, label in enumerate(['0=Less than 1h/day', '1=Around 1h/day', '2=Around 2hs/day', '3=More than 3hs/day']):\n    handles.append(plt.Line2D([0], [0], color=palette[i], lw=4))  # Create a line for each color\n\nplt.legend(handles, \n           ['0=Less than 1h/day', '1=Around 1h/day', '2=Around 2hs/day', '3=More than 3hs/day'], \n           title='Daily Internet Usage')\n\nplt.tight_layout() \nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:39.858538Z","iopub.execute_input":"2024-10-11T16:47:39.859051Z","iopub.status.idle":"2024-10-11T16:47:40.673721Z","shell.execute_reply.started":"2024-10-11T16:47:39.859004Z","shell.execute_reply":"2024-10-11T16:47:40.672652Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sleep Quality vs internet usage","metadata":{}},{"cell_type":"code","source":"mean_sleep_time = data.groupby('PreInt_EduHx-computerinternet_hoursday')['SDS-SDS_Total_T'].mean()\n\nplt.figure(figsize=(10, 8))\nplt.plot(mean_sleep_time.index, mean_sleep_time.values, marker='o', linestyle='-', color='blue')\n\n# Customize the plot\nplt.xlabel('Internet Usage Time (hours/day)')\nplt.ylabel('Average Sleep Time (hours)')\nplt.title('Average Sleep Time vs Internet Usage Time')\n\ncustom_ticks = [ 45, 50, 55, 60, 65, 70,75]  # Define your custom tick values\nplt.yticks(custom_ticks)\n\nplt.grid(True)\n\n# Show the plot\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:40.675290Z","iopub.execute_input":"2024-10-11T16:47:40.675639Z","iopub.status.idle":"2024-10-11T16:47:40.989411Z","shell.execute_reply.started":"2024-10-11T16:47:40.675579Z","shell.execute_reply":"2024-10-11T16:47:40.988421Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Behavioural influence ","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n\nplt.figure(figsize=(12, 6))\nsns.boxplot(x='PCIAT-PCIAT_06', y='Basic_Demos-Age', data=data, palette='pastel')\nplt.title('Age Distribution for Q: How often do your childs grades suffer because of the amount of time he or she spends online?')\nplt.xlabel('Response')\nplt.ylabel('Age')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:40.990868Z","iopub.execute_input":"2024-10-11T16:47:40.991313Z","iopub.status.idle":"2024-10-11T16:47:41.354429Z","shell.execute_reply.started":"2024-10-11T16:47:40.991264Z","shell.execute_reply":"2024-10-11T16:47:41.353383Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# List of questions to analyze\nquestions = [\n    'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', \n    'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06',\n    'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09',\n    'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12',\n    'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15',\n    'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18',\n    'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20'\n]\n\nplt.figure(figsize=(15, 40))  # Adjust the size for better visibility\n\n# Create a box plot for each question\nfor i, question in enumerate(questions):\n    plt.subplot(len(questions), 1, i + 1)  # Create subplots\n    sns.boxplot(x=question, y='Basic_Demos-Age', data=data, palette='pastel')\n    plt.title(f'Age Distribution for {question}')\n    plt.xlabel('Response')\n    plt.ylabel('Age')\n\nplt.tight_layout()  # Adjust layout to prevent overlap\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:41.355615Z","iopub.execute_input":"2024-10-11T16:47:41.355929Z","iopub.status.idle":"2024-10-11T16:47:46.561305Z","shell.execute_reply.started":"2024-10-11T16:47:41.355896Z","shell.execute_reply":"2024-10-11T16:47:46.559965Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"markdown","source":"## Handling Missing Values ","metadata":{}},{"cell_type":"markdown","source":"### Drop columns with null>50","metadata":{}},{"cell_type":"code","source":"\ncol = ['CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n       'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season',\n       'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']\ncol.extend( \n        ['Physical-Waist_Circumference','Fitness_Endurance-Max_Stage',\n         'Fitness_Endurance-Time_Mins','Fitness_Endurance-Time_Sec','FGC-FGC_GSND',\n         'FGC-FGC_GSND_Zone','FGC-FGC_GSD','FGC-FGC_GSD_Zone','BIA-BIA_Activity_Level_num',\n         'BIA-BIA_BMC','BIA-BIA_BMI','BIA-BIA_BMR','BIA-BIA_DEE','BIA-BIA_ECW','BIA-BIA_FFM',\n         'BIA-BIA_FFMI','BIA-BIA_Fat','BIA-BIA_Frame_num','BIA-BIA_ICW','BIA-BIA_LDM','BIA-BIA_LST',\n         'BIA-BIA_SMM','BIA-BIA_TBW','PAQ_A-PAQ_A_Total','PAQ_C-PAQ_C_Total'])\nlen(col)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:46.562579Z","iopub.execute_input":"2024-10-11T16:47:46.562944Z","iopub.status.idle":"2024-10-11T16:47:46.571133Z","shell.execute_reply.started":"2024-10-11T16:47:46.562907Z","shell.execute_reply":"2024-10-11T16:47:46.570045Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop(col , axis =1 ,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:46.572440Z","iopub.execute_input":"2024-10-11T16:47:46.572804Z","iopub.status.idle":"2024-10-11T16:47:46.593631Z","shell.execute_reply.started":"2024-10-11T16:47:46.572768Z","shell.execute_reply":"2024-10-11T16:47:46.592584Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:46.597211Z","iopub.execute_input":"2024-10-11T16:47:46.597553Z","iopub.status.idle":"2024-10-11T16:47:46.606706Z","shell.execute_reply.started":"2024-10-11T16:47:46.597519Z","shell.execute_reply":"2024-10-11T16:47:46.605797Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned= data.copy()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:46.607906Z","iopub.execute_input":"2024-10-11T16:47:46.608249Z","iopub.status.idle":"2024-10-11T16:47:46.619478Z","shell.execute_reply.started":"2024-10-11T16:47:46.608212Z","shell.execute_reply":"2024-10-11T16:47:46.618442Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:46.620718Z","iopub.execute_input":"2024-10-11T16:47:46.621018Z","iopub.status.idle":"2024-10-11T16:47:46.630321Z","shell.execute_reply.started":"2024-10-11T16:47:46.620987Z","shell.execute_reply":"2024-10-11T16:47:46.629259Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Filling Missing Values in first 46 columns with Knn imputer","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\ndata_cleaned['Basic_Demos-Enroll_Season'] = label_encoder.fit_transform(data_cleaned['Basic_Demos-Enroll_Season'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:46.631585Z","iopub.execute_input":"2024-10-11T16:47:46.631920Z","iopub.status.idle":"2024-10-11T16:47:46.643041Z","shell.execute_reply.started":"2024-10-11T16:47:46.631887Z","shell.execute_reply":"2024-10-11T16:47:46.642065Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\nimputer = KNNImputer(n_neighbors=5)\ndata_cleaned.iloc[:, :45] = imputer.fit_transform(data_cleaned.iloc[:, :45])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:46.644234Z","iopub.execute_input":"2024-10-11T16:47:46.644536Z","iopub.status.idle":"2024-10-11T16:47:51.246806Z","shell.execute_reply.started":"2024-10-11T16:47:46.644504Z","shell.execute_reply":"2024-10-11T16:47:51.245929Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\ndata_cleaned['Severity'] = label_encoder.fit_transform(data_cleaned['Severity'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:51.247935Z","iopub.execute_input":"2024-10-11T16:47:51.248240Z","iopub.status.idle":"2024-10-11T16:47:51.257528Z","shell.execute_reply.started":"2024-10-11T16:47:51.248208Z","shell.execute_reply":"2024-10-11T16:47:51.256614Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Filling missing values in target with Kmeans","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nimport numpy as np\nimport pandas as pd\n\ndef impute_with_kmeans(df, categorical_columns, n_clusters=4):\n    # Fill missing values temporarily with the mode (or any placeholder)\n    df_temp = df.copy()\n    for col in categorical_columns:\n        df_temp[col].fillna(df_temp[col].mode()[0], inplace=True)\n\n    # Perform KMeans clustering after filling missing values\n    kmeans = KMeans(n_clusters=n_clusters, random_state=0)\n    cluster_labels = kmeans.fit_predict(df_temp)\n\n    # Impute missing values within each cluster\n    for col in categorical_columns:\n        for cluster in np.unique(cluster_labels):\n            mask = (cluster_labels == cluster) & df[col].isna()\n            most_frequent = df.loc[cluster_labels == cluster, col].mode()[0]\n            df.loc[mask, col] = most_frequent\n\n    return df\n\n# Apply KMeans-based imputation\ndata_cleaned = impute_with_kmeans(data_cleaned, ['sii'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:51.258831Z","iopub.execute_input":"2024-10-11T16:47:51.259228Z","iopub.status.idle":"2024-10-11T16:47:51.946476Z","shell.execute_reply.started":"2024-10-11T16:47:51.259182Z","shell.execute_reply":"2024-10-11T16:47:51.944881Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned['sii'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:51.948319Z","iopub.execute_input":"2024-10-11T16:47:51.952406Z","iopub.status.idle":"2024-10-11T16:47:51.963292Z","shell.execute_reply.started":"2024-10-11T16:47:51.952365Z","shell.execute_reply":"2024-10-11T16:47:51.962204Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:51.964925Z","iopub.execute_input":"2024-10-11T16:47:51.965333Z","iopub.status.idle":"2024-10-11T16:47:51.997075Z","shell.execute_reply.started":"2024-10-11T16:47:51.965286Z","shell.execute_reply":"2024-10-11T16:47:51.995912Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:51.998073Z","iopub.execute_input":"2024-10-11T16:47:51.998400Z","iopub.status.idle":"2024-10-11T16:47:52.012782Z","shell.execute_reply.started":"2024-10-11T16:47:51.998357Z","shell.execute_reply":"2024-10-11T16:47:52.011684Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handling outliers ","metadata":{}},{"cell_type":"code","source":"outlier_indices = detect_outliers_iqr(data_cleaned)\nprint(\"Total outliers detected:\", len(set(outlier_indices)))","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:52.013987Z","iopub.execute_input":"2024-10-11T16:47:52.014298Z","iopub.status.idle":"2024-10-11T16:47:52.115786Z","shell.execute_reply.started":"2024-10-11T16:47:52.014265Z","shell.execute_reply":"2024-10-11T16:47:52.114675Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for column in data_cleaned.columns:\n    plt.figure(figsize=(8, 3))  \n    sns.boxplot(data=data_cleaned[[column]], orient='h', palette=\"Set2\")\n    \n    plt.title(f'Box Plot for {column} with Outliers', fontsize=16)\n    plt.ylabel(column, fontsize=14)\n    plt.xlabel('Values', fontsize=14)\n    \n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:47:52.117269Z","iopub.execute_input":"2024-10-11T16:47:52.117746Z","iopub.status.idle":"2024-10-11T16:48:06.846122Z","shell.execute_reply.started":"2024-10-11T16:47:52.117696Z","shell.execute_reply":"2024-10-11T16:48:06.845101Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate IQR for the column 'CGAS-CGAS_Score'\nQ1 = data_cleaned['CGAS-CGAS_Score'].quantile(0.25)\nQ3 = data_cleaned['CGAS-CGAS_Score'].quantile(0.75)\nIQR = Q3 - Q1\n\n# Define outlier boundaries\nlower_bound = Q1 - 1.5 * IQR\nupper_bound = Q3 + 1.5 * IQR\n\n# Filter the entire DataFrame by keeping only rows where 'CGAS-CGAS_Score' is within bounds\ndata_copy = data_cleaned[(data_cleaned['CGAS-CGAS_Score'] >= lower_bound) & (data_cleaned['CGAS-CGAS_Score'] <= upper_bound)]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:06.847434Z","iopub.execute_input":"2024-10-11T16:48:06.847771Z","iopub.status.idle":"2024-10-11T16:48:06.857719Z","shell.execute_reply.started":"2024-10-11T16:48:06.847735Z","shell.execute_reply":"2024-10-11T16:48:06.856645Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"## BMI To Heart rate Ratio","metadata":{}},{"cell_type":"code","source":"data_copy['HeartRate_BMI'] = data_copy['Physical-HeartRate'] * data_copy['Physical-BMI']","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:06.858917Z","iopub.execute_input":"2024-10-11T16:48:06.859242Z","iopub.status.idle":"2024-10-11T16:48:06.871300Z","shell.execute_reply.started":"2024-10-11T16:48:06.859209Z","shell.execute_reply":"2024-10-11T16:48:06.870359Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Max Heart rate due to age","metadata":{}},{"cell_type":"code","source":"data_copy['HRmax'] = 220 - data_copy['Basic_Demos-Age']  # Estimate HRmax based on age","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:06.872459Z","iopub.execute_input":"2024-10-11T16:48:06.872792Z","iopub.status.idle":"2024-10-11T16:48:06.890509Z","shell.execute_reply.started":"2024-10-11T16:48:06.872757Z","shell.execute_reply":"2024-10-11T16:48:06.889644Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Addation Feature ","metadata":{}},{"cell_type":"code","source":"def categorize_bmi(bmi):\n    if bmi <= 18.5:\n        return 0\n    elif bmi <= 24.9:\n        return 1\n    elif bmi <= 29.9:\n        return 2\n    elif bmi <= 40:\n        return 3\n    else:\n        return 4  \n\ndata_copy['BMI_Category'] = data_copy['Physical-BMI'].apply(categorize_bmi)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:48:06.891683Z","iopub.execute_input":"2024-10-11T16:48:06.892000Z","iopub.status.idle":"2024-10-11T16:48:06.905324Z","shell.execute_reply.started":"2024-10-11T16:48:06.891966Z","shell.execute_reply":"2024-10-11T16:48:06.904323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pciat_columns = [f'PCIAT-PCIAT_{i:02d}' for i in range(1, 21)]\ndata_copy['PCIAT_Time_Management'] = data_copy[pciat_columns[:5]].mean(axis=1)\ndata_copy['PCIAT_Withdrawal_Symptoms'] = data_copy[pciat_columns[5:10]].mean(axis=1)\ndata_copy['PCIAT_Neglect_Social_Life'] = data_copy[pciat_columns[10:15]].mean(axis=1)\ndata_copy['PCIAT_Lack_Control'] = data_copy[pciat_columns[15:]].mean(axis=1)\n\n\n\ndata_copy['PCIAT_mean'] = data_copy[pciat_columns].mean(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:48:06.906720Z","iopub.execute_input":"2024-10-11T16:48:06.907147Z","iopub.status.idle":"2024-10-11T16:48:06.928413Z","shell.execute_reply.started":"2024-10-11T16:48:06.907101Z","shell.execute_reply":"2024-10-11T16:48:06.927353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_pciat(score):\n    if score <= 20:\n         return 0\n    elif score <= 49:\n          return 1\n    elif score <= 79:\n         return 2\n    else:\n         return 3\n    \ndata_copy['PCIAT_Category'] = data_copy['PCIAT-PCIAT_Total'].apply(categorize_pciat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:48:06.929776Z","iopub.execute_input":"2024-10-11T16:48:06.930298Z","iopub.status.idle":"2024-10-11T16:48:06.939036Z","shell.execute_reply.started":"2024-10-11T16:48:06.930248Z","shell.execute_reply":"2024-10-11T16:48:06.937919Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_sds(score):\n    if pd.isna(score) or score < 0:\n        return np.nan\n    elif score <= 20:\n        return 0\n    elif score <= 40:\n        return 1\n    elif score <= 60:\n        return 2\n    elif score <= 80:\n        return 3\n    else:\n        return 4  \n\ndata_copy['SDS_Severity'] = data_copy['SDS-SDS_Total_Raw'].apply(categorize_sds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:48:06.940446Z","iopub.execute_input":"2024-10-11T16:48:06.940872Z","iopub.status.idle":"2024-10-11T16:48:06.955369Z","shell.execute_reply.started":"2024-10-11T16:48:06.940824Z","shell.execute_reply":"2024-10-11T16:48:06.954381Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['Sleep_Quality_Index'] = (data_copy['SDS-SDS_Total_T'] - data_copy['SDS-SDS_Total_T'].min()) / (data_copy['SDS-SDS_Total_T'].max() - data_copy['SDS-SDS_Total_T'].min())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:48:06.956743Z","iopub.execute_input":"2024-10-11T16:48:06.957615Z","iopub.status.idle":"2024-10-11T16:48:06.967210Z","shell.execute_reply.started":"2024-10-11T16:48:06.957545Z","shell.execute_reply":"2024-10-11T16:48:06.966208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['Physical_Health_Index'] = ((data_copy['Physical-BMI'] - data_copy['Physical-BMI'].mean()) / data_copy['Physical-BMI'].std() + (data_copy['Physical-Systolic_BP'] - data_copy['Physical-Systolic_BP'].mean()) / data_copy['Physical-Systolic_BP'].std() + (data_copy['Physical-HeartRate'] - data_copy['Physical-HeartRate'].mean()) / data_copy['Physical-HeartRate'].std()) / 3\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:48:06.968546Z","iopub.execute_input":"2024-10-11T16:48:06.969210Z","iopub.status.idle":"2024-10-11T16:48:06.981534Z","shell.execute_reply.started":"2024-10-11T16:48:06.969162Z","shell.execute_reply":"2024-10-11T16:48:06.980539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['Sleep_Quality_Index'] = (data_copy['SDS-SDS_Total_T'] - data_copy['SDS-SDS_Total_T'].min()) / (data_copy['SDS-SDS_Total_T'].max() - data_copy['SDS-SDS_Total_T'].min())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:48:06.982890Z","iopub.execute_input":"2024-10-11T16:48:06.983307Z","iopub.status.idle":"2024-10-11T16:48:06.995607Z","shell.execute_reply.started":"2024-10-11T16:48:06.983261Z","shell.execute_reply":"2024-10-11T16:48:06.994628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fgc_columns = ['FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_TL']\n\ndata_copy['Overall_Fitness_Score'] = data_copy[fgc_columns].mean(axis=1)\ndata_copy['Internet_Usage_Score'] = data_copy['PCIAT-PCIAT_Total'] / 100\ndata_copy['Physical_Activity_Score'] = data_copy['Overall_Fitness_Score'] / data_copy['Overall_Fitness_Score'].max()\ndata_copy['Lifestyle_Score'] = ((1 - data_copy['Internet_Usage_Score']) +  data_copy['Physical_Activity_Score'] + (1 - data_copy['Sleep_Quality_Index'])) / 3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:48:06.996740Z","iopub.execute_input":"2024-10-11T16:48:06.997149Z","iopub.status.idle":"2024-10-11T16:48:07.011958Z","shell.execute_reply.started":"2024-10-11T16:48:06.997115Z","shell.execute_reply":"2024-10-11T16:48:07.010836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy = data_copy.drop(columns=['SDS-SDS_Total_T', 'SDS-SDS_Total_Raw', 'PCIAT-PCIAT_Total', 'Physical-Height', 'Physical-Weight', 'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_TL', 'FGC-FGC_CU_Zone', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_TL_Zone', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20'])\n\n\n\ndata_copy = data_copy.drop(columns=['PCIAT_Time_Management', 'PCIAT_Withdrawal_Symptoms', 'PCIAT_Neglect_Social_Life', 'PCIAT_Lack_Control',  'Basic_Demos-Age', 'Physical-BMI'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:48:07.013263Z","iopub.execute_input":"2024-10-11T16:48:07.013611Z","iopub.status.idle":"2024-10-11T16:48:07.024524Z","shell.execute_reply.started":"2024-10-11T16:48:07.013556Z","shell.execute_reply":"2024-10-11T16:48:07.023528Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Body Strength & Flexibility","metadata":{}},{"cell_type":"code","source":"# data_copy['BodyStrength&Flexibility'] = data_copy['FGC-FGC_CU'] + data_copy['FGC-FGC_PU'] + data_copy['FGC-FGC_SRL'] + data_copy['FGC-FGC_SRR'] + data_copy['FGC-FGC_TL']\n# data_copy['BodyStrength&Flexibility_class'] = data_copy['FGC-FGC_CU_Zone'] + data_copy['FGC-FGC_PU_Zone'] + data_copy['FGC-FGC_SRL_Zone'] + data_copy['FGC-FGC_SRR_Zone'] + data_copy['FGC-FGC_TL_Zone']\n\n# data_copy = data_copy.drop(['FGC-FGC_CU','FGC-FGC_PU','FGC-FGC_SRL','FGC-FGC_SRR','FGC-FGC_TL'], axis=1)\n\n# data_copy = data_copy.drop(['FGC-FGC_CU_Zone','FGC-FGC_PU_Zone','FGC-FGC_SRL_Zone','FGC-FGC_SRR_Zone','FGC-FGC_TL_Zone'], axis=1)                          \n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:07.025890Z","iopub.execute_input":"2024-10-11T16:48:07.026224Z","iopub.status.idle":"2024-10-11T16:48:07.033765Z","shell.execute_reply.started":"2024-10-11T16:48:07.026189Z","shell.execute_reply":"2024-10-11T16:48:07.032782Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:07.034998Z","iopub.execute_input":"2024-10-11T16:48:07.035324Z","iopub.status.idle":"2024-10-11T16:48:07.050793Z","shell.execute_reply.started":"2024-10-11T16:48:07.035289Z","shell.execute_reply":"2024-10-11T16:48:07.049774Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Transfomation and scaling","metadata":{}},{"cell_type":"markdown","source":"## Data Splitting ","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = data_copy.drop('sii', axis=1)  \ny = data_copy['sii']  \nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)\nprint(X_train.shape, X_test.shape, y_train.shape, y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:07.051948Z","iopub.execute_input":"2024-10-11T16:48:07.052272Z","iopub.status.idle":"2024-10-11T16:48:07.067180Z","shell.execute_reply.started":"2024-10-11T16:48:07.052238Z","shell.execute_reply":"2024-10-11T16:48:07.065607Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handling imbalanced data ","metadata":{}},{"cell_type":"code","source":"from collections import Counter\nfrom imblearn.over_sampling import SMOTE\n\n# Check the class distribution before applying SMOTE\nclass_distribution_before = Counter(y_train)\nprint(\"Class distribution before SMOTE:\", class_distribution_before)\n\n# Initialize SMOTE with specified parameters\nsmote = SMOTE(sampling_strategy='auto', random_state=42)\n\n# Apply SMOTE to the training data\nX_train_resampled, y_train_resampled = smote.fit_resample(X_train, y_train)\n\n# Check the class distribution after applying SMOTE\nclass_distribution_after = Counter(y_train_resampled)\nprint(\"Class distribution after SMOTE:\", class_distribution_after)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:07.068509Z","iopub.execute_input":"2024-10-11T16:48:07.068866Z","iopub.status.idle":"2024-10-11T16:48:07.097445Z","shell.execute_reply.started":"2024-10-11T16:48:07.068822Z","shell.execute_reply":"2024-10-11T16:48:07.096357Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train = y_train_resampled","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:48:07.098707Z","iopub.execute_input":"2024-10-11T16:48:07.099051Z","iopub.status.idle":"2024-10-11T16:48:07.103583Z","shell.execute_reply.started":"2024-10-11T16:48:07.099015Z","shell.execute_reply":"2024-10-11T16:48:07.102554Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Scaling","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import RobustScaler\nscaler = RobustScaler()  # or StandardScaler()\n\n# Fit the scaler on X_train and transform both X_train and X_test\nX_train = pd.DataFrame(scaler.fit_transform(X_train_resampled), columns=X_train_resampled.columns)\nX_test = pd.DataFrame(scaler.transform(X_test), columns=X_test.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:07.104700Z","iopub.execute_input":"2024-10-11T16:48:07.104998Z","iopub.status.idle":"2024-10-11T16:48:07.132652Z","shell.execute_reply.started":"2024-10-11T16:48:07.104956Z","shell.execute_reply":"2024-10-11T16:48:07.131651Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.shape\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:07.133971Z","iopub.execute_input":"2024-10-11T16:48:07.134292Z","iopub.status.idle":"2024-10-11T16:48:07.140871Z","shell.execute_reply.started":"2024-10-11T16:48:07.134259Z","shell.execute_reply":"2024-10-11T16:48:07.139700Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:07.142084Z","iopub.execute_input":"2024-10-11T16:48:07.142389Z","iopub.status.idle":"2024-10-11T16:48:07.152280Z","shell.execute_reply.started":"2024-10-11T16:48:07.142349Z","shell.execute_reply":"2024-10-11T16:48:07.151328Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PCA","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA \npca = PCA(n_components= 20 )  # Reduce to 2 components\nX_train = pca.fit_transform(X_train)\nX_test = pca.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:48:07.153786Z","iopub.execute_input":"2024-10-11T16:48:07.154143Z","iopub.status.idle":"2024-10-11T16:48:07.920644Z","shell.execute_reply.started":"2024-10-11T16:48:07.154109Z","shell.execute_reply":"2024-10-11T16:48:07.918369Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modeling ","metadata":{}},{"cell_type":"code","source":"pip install lazypredict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:48:07.928978Z","iopub.execute_input":"2024-10-11T16:48:07.929322Z","iopub.status.idle":"2024-10-11T16:48:18.849942Z","shell.execute_reply.started":"2024-10-11T16:48:07.929286Z","shell.execute_reply":"2024-10-11T16:48:18.848528Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lazypredict.Supervised import LazyClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.datasets import load_iris\nimport pandas as pd\n\n\n\nclf = LazyClassifier(verbose=0, ignore_warnings=True, custom_metric=None)\n\nmodels, predictions = clf.fit(X_train, X_test, y_train, y_test)\n\nprint(models)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:48:18.851648Z","iopub.execute_input":"2024-10-11T16:48:18.852019Z","iopub.status.idle":"2024-10-11T16:48:48.557550Z","shell.execute_reply.started":"2024-10-11T16:48:18.851977Z","shell.execute_reply":"2024-10-11T16:48:48.556470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result1, result2, result3 = [], [], [] ","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:48.559201Z","iopub.execute_input":"2024-10-11T16:48:48.559658Z","iopub.status.idle":"2024-10-11T16:48:48.564989Z","shell.execute_reply.started":"2024-10-11T16:48:48.559608Z","shell.execute_reply":"2024-10-11T16:48:48.563891Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_f1_scores = []\ntest_f1_scores = []\nmodel_names = []\n\ndef modeling(model, model_name):\n    # Fit the model using the scaled resampled training data\n    global train_f1_scores, test_f1_scores, model_names\n\n    model.fit(X_train, y_train_resampled)  # Use y_train_resampled here\n    \n    # Predictions\n    train_pred = model.predict(X_train)    # Predictions on the scaled resampled train set\n    test_pred = model.predict(X_test)      # Predictions on the scaled original test set\n    \n    # Calculate metrics with specified average\n    train_accuracy = accuracy_score(y_train_resampled, train_pred) * 100\n    train_recall = recall_score(y_train_resampled, train_pred, average='weighted') * 100\n    train_f1_score = f1_score(y_train_resampled, train_pred, average='weighted') * 100\n    \n    test_accuracy = accuracy_score(y_test, test_pred) * 100\n    test_recall = recall_score(y_test, test_pred, average='weighted') * 100\n    test_f1_score = f1_score(y_test, test_pred, average='weighted') * 100\n\n\n    train_f1_scores.append(train_f1_score)\n    test_f1_scores.append(test_f1_score)\n    model_names.append(model_name)\n\n    \n    # Append results\n    result1.append(test_accuracy)\n    result2.append(test_recall)\n    result3.append(test_f1_score)\n    \n    print(\"Classification Report for Test Data:\")\n    print(classification_report(y_test, test_pred))\n    \n    print(\"\\nClassification Report for Scaled Resampled Train Data:\")\n    print(classification_report(y_train_resampled, train_pred))  # Use y_train_resampled here\n    \n    # Accuracy, Recall, and F1 Scores\n    print(f'Training Accuracy: {train_accuracy}, Train Recall: {train_recall}, Train F1: {train_f1_score}')\n    print(f'Test Accuracy: {test_accuracy}, Test Recall: {test_recall}, Test F1: {test_f1_score}')\n    \n    # Confusion matrix\n    cm = confusion_matrix(y_test, test_pred)\n    sns.heatmap(cm, annot=True, fmt='0.2f', cmap='YlGnBu', linewidths=1)\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n    plt.title('Confusion Matrix')\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:48.566661Z","iopub.execute_input":"2024-10-11T16:48:48.567086Z","iopub.status.idle":"2024-10-11T16:48:48.579095Z","shell.execute_reply.started":"2024-10-11T16:48:48.567037Z","shell.execute_reply":"2024-10-11T16:48:48.577889Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression  \nlogistic_regression = LogisticRegression()\nmodeling(logistic_regression, 'Logistic Regression')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:48.580403Z","iopub.execute_input":"2024-10-11T16:48:48.580796Z","iopub.status.idle":"2024-10-11T16:48:49.615054Z","shell.execute_reply.started":"2024-10-11T16:48:48.580760Z","shell.execute_reply":"2024-10-11T16:48:49.614018Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SVM = SVC()\nmodeling(SVM, 'SVM')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:49.616946Z","iopub.execute_input":"2024-10-11T16:48:49.617304Z","iopub.status.idle":"2024-10-11T16:48:50.862546Z","shell.execute_reply.started":"2024-10-11T16:48:49.617268Z","shell.execute_reply":"2024-10-11T16:48:50.861535Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decision_tree_classifier = DecisionTreeClassifier()\nmodeling(decision_tree_classifier, 'Decision Tree')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:50.864054Z","iopub.execute_input":"2024-10-11T16:48:50.864477Z","iopub.status.idle":"2024-10-11T16:48:51.473919Z","shell.execute_reply.started":"2024-10-11T16:48:50.864429Z","shell.execute_reply":"2024-10-11T16:48:51.472856Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decision_tree_classifier = DecisionTreeClassifier( max_depth= 7 , min_samples_split= 7)\nmodeling(decision_tree_classifier, 'Decision Tree Tuned')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:51.475473Z","iopub.execute_input":"2024-10-11T16:48:51.475933Z","iopub.status.idle":"2024-10-11T16:48:52.005409Z","shell.execute_reply.started":"2024-10-11T16:48:51.475884Z","shell.execute_reply":"2024-10-11T16:48:52.004337Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sgd_classifier = SGDClassifier(max_iter = 500) \nmodeling(sgd_classifier, 'SGD')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:52.016435Z","iopub.execute_input":"2024-10-11T16:48:52.016820Z","iopub.status.idle":"2024-10-11T16:48:52.579478Z","shell.execute_reply.started":"2024-10-11T16:48:52.016783Z","shell.execute_reply":"2024-10-11T16:48:52.578214Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nknn = KNeighborsClassifier()\nmodeling(knn, 'KNN Classifier')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:52.580871Z","iopub.execute_input":"2024-10-11T16:48:52.581212Z","iopub.status.idle":"2024-10-11T16:48:53.595558Z","shell.execute_reply.started":"2024-10-11T16:48:52.581176Z","shell.execute_reply":"2024-10-11T16:48:53.594535Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n\n# Create an instance of GaussianNB\nnaive_bayes = GaussianNB()\n\n# Call your modeling function\nmodeling(naive_bayes, 'Naive Bayes')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:53.596911Z","iopub.execute_input":"2024-10-11T16:48:53.597253Z","iopub.status.idle":"2024-10-11T16:48:54.001026Z","shell.execute_reply.started":"2024-10-11T16:48:53.597217Z","shell.execute_reply":"2024-10-11T16:48:53.999893Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nrandom_forest = RandomForestClassifier(random_state=1)\nmodeling(random_forest, 'Random Forest')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:54.002243Z","iopub.execute_input":"2024-10-11T16:48:54.002550Z","iopub.status.idle":"2024-10-11T16:48:57.065606Z","shell.execute_reply.started":"2024-10-11T16:48:54.002517Z","shell.execute_reply":"2024-10-11T16:48:57.064548Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"adaBosster=AdaBoostClassifier( n_estimators=50)\nmodeling(adaBosster, 'AdaBoost')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:57.067112Z","iopub.execute_input":"2024-10-11T16:48:57.067535Z","iopub.status.idle":"2024-10-11T16:48:58.962278Z","shell.execute_reply.started":"2024-10-11T16:48:57.067487Z","shell.execute_reply":"2024-10-11T16:48:58.961227Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingClassifier\n\ngradient=GradientBoostingClassifier()\nmodeling(gradient, 'Gradient Boost')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:48:58.963616Z","iopub.execute_input":"2024-10-11T16:48:58.963962Z","iopub.status.idle":"2024-10-11T16:49:20.708846Z","shell.execute_reply.started":"2024-10-11T16:48:58.963921Z","shell.execute_reply":"2024-10-11T16:49:20.707780Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_f1_scores():\n    plt.figure(figsize=(12, 6))\n    x = range(len(model_names))\n    \n    plt.plot(x, train_f1_scores, 'bo-', label='Training F1 Score')\n    plt.plot(x, test_f1_scores, 'ro-', label='Testing F1 Score')\n    \n    plt.xlabel('Models')\n    plt.ylabel('F1 Score (Weighted)')\n    plt.title('Training and Testing F1 Scores for Different Models')\n    plt.xticks(x, model_names, rotation=45, ha='right')\n    plt.legend()\n    plt.tight_layout()\n    plt.show()\n\ndef plot_f1_scores1():\n    plt.figure(figsize=(12, 6))\n    \n    x = np.arange(len(model_names))  # the label locations\n    width = 0.35  # the width of the bars\n    \n    # Create the bars\n    rects1 = plt.bar(x - width/2, train_f1_scores, width, label='Train', color='blue', alpha=0.7)\n    rects2 = plt.bar(x + width/2, test_f1_scores, width, label='Test', color='red', alpha=0.7)\n\n    # Add some text for labels, title and custom x-axis tick labels, etc.\n    plt.ylabel('F1 Score (Weighted)')\n    plt.title('F1 Scores for Different Models (Training and Testing)')\n    plt.xticks(x, model_names, rotation=45, ha='right')\n    plt.legend()\n\n    # Add value labels on the bars\n    def autolabel(rects):\n        for rect in rects:\n            height = rect.get_height()\n            plt.annotate(f'{height:.1f}',\n                        xy=(rect.get_x() + rect.get_width() / 2, height),\n                        xytext=(0, 3),  # 3 points vertical offset\n                        textcoords=\"offset points\",\n                        ha='center', va='bottom')\n\n    autolabel(rects1)\n    autolabel(rects2)\n\n    plt.tight_layout()\n    plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:54:21.871633Z","iopub.execute_input":"2024-10-11T16:54:21.872091Z","iopub.status.idle":"2024-10-11T16:54:21.883015Z","shell.execute_reply.started":"2024-10-11T16:54:21.872051Z","shell.execute_reply":"2024-10-11T16:54:21.881929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_f1_scores()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:49:20.718293Z","iopub.execute_input":"2024-10-11T16:49:20.718654Z","iopub.status.idle":"2024-10-11T16:49:21.222705Z","shell.execute_reply.started":"2024-10-11T16:49:20.718603Z","shell.execute_reply":"2024-10-11T16:49:21.221571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plot_f1_scores1()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:54:24.928888Z","iopub.execute_input":"2024-10-11T16:54:24.929307Z","iopub.status.idle":"2024-10-11T16:54:25.543820Z","shell.execute_reply.started":"2024-10-11T16:54:24.929268Z","shell.execute_reply":"2024-10-11T16:54:25.542728Z"}},"outputs":[],"execution_count":null}]}