{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt \nimport pandas as pd \nimport plotly.express as px\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport math\nfrom plotly.subplots import make_subplots\nimport numpy as np\nfrom numpy import linalg as LA\nimport plotly.express as px\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\nimport warnings\n\nwarnings.filterwarnings('ignore')\n\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nfrom sklearn.svm import SVC\nimport numpy as np\nimport matplotlib.pyplot as plt \nimport pandas as pd \nimport plotly.express as px\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport math\nfrom plotly.subplots import make_subplots\nimport numpy as np\nfrom numpy import linalg as LA\nimport plotly.express as px\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\n\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import recall_score, f1_score, accuracy_score\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import (accuracy_score, recall_score, f1_score, \n                             classification_report, confusion_matrix)\nfrom sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:37.001192Z","iopub.execute_input":"2024-10-11T16:19:37.001681Z","iopub.status.idle":"2024-10-11T16:19:37.016701Z","shell.execute_reply.started":"2024-10-11T16:19:37.001627Z","shell.execute_reply":"2024-10-11T16:19:37.015379Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Exploration ","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:37.185365Z","iopub.execute_input":"2024-10-11T16:19:37.185778Z","iopub.status.idle":"2024-10-11T16:19:37.242104Z","shell.execute_reply.started":"2024-10-11T16:19:37.185738Z","shell.execute_reply":"2024-10-11T16:19:37.240974Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:37.586346Z","iopub.execute_input":"2024-10-11T16:19:37.586822Z","iopub.status.idle":"2024-10-11T16:19:37.597736Z","shell.execute_reply.started":"2024-10-11T16:19:37.586777Z","shell.execute_reply":"2024-10-11T16:19:37.596495Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:38.020290Z","iopub.execute_input":"2024-10-11T16:19:38.020718Z","iopub.status.idle":"2024-10-11T16:19:38.076182Z","shell.execute_reply.started":"2024-10-11T16:19:38.020678Z","shell.execute_reply":"2024-10-11T16:19:38.075206Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:38.427282Z","iopub.execute_input":"2024-10-11T16:19:38.427735Z","iopub.status.idle":"2024-10-11T16:19:38.453856Z","shell.execute_reply.started":"2024-10-11T16:19:38.427691Z","shell.execute_reply":"2024-10-11T16:19:38.452644Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop('id', axis=1 , inplace = True)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:38.663260Z","iopub.execute_input":"2024-10-11T16:19:38.664481Z","iopub.status.idle":"2024-10-11T16:19:38.672064Z","shell.execute_reply.started":"2024-10-11T16:19:38.664429Z","shell.execute_reply":"2024-10-11T16:19:38.670791Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:38.990708Z","iopub.execute_input":"2024-10-11T16:19:38.991133Z","iopub.status.idle":"2024-10-11T16:19:39.026412Z","shell.execute_reply.started":"2024-10-11T16:19:38.991091Z","shell.execute_reply":"2024-10-11T16:19:39.025121Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop_duplicates(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:39.155557Z","iopub.execute_input":"2024-10-11T16:19:39.156000Z","iopub.status.idle":"2024-10-11T16:19:39.186758Z","shell.execute_reply.started":"2024-10-11T16:19:39.155956Z","shell.execute_reply":"2024-10-11T16:19:39.185596Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:39.471124Z","iopub.execute_input":"2024-10-11T16:19:39.471607Z","iopub.status.idle":"2024-10-11T16:19:39.510632Z","shell.execute_reply.started":"2024-10-11T16:19:39.471561Z","shell.execute_reply":"2024-10-11T16:19:39.509474Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:39.905202Z","iopub.execute_input":"2024-10-11T16:19:39.906213Z","iopub.status.idle":"2024-10-11T16:19:40.111750Z","shell.execute_reply.started":"2024-10-11T16:19:39.906167Z","shell.execute_reply":"2024-10-11T16:19:40.110661Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:40.155288Z","iopub.execute_input":"2024-10-11T16:19:40.156098Z","iopub.status.idle":"2024-10-11T16:19:40.171462Z","shell.execute_reply.started":"2024-10-11T16:19:40.156050Z","shell.execute_reply":"2024-10-11T16:19:40.170496Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_cols = [col for col in data.columns if data[col].dtype != 'O']\ncat_cols = [col for col in data.columns if col not in num_cols]\nprint(f'Numerical columns: {num_cols}')\nprint(\"-------------------\")\nprint(f'Categorical columns: {cat_cols}')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:40.470077Z","iopub.execute_input":"2024-10-11T16:19:40.470973Z","iopub.status.idle":"2024-10-11T16:19:40.479542Z","shell.execute_reply.started":"2024-10-11T16:19:40.470924Z","shell.execute_reply":"2024-10-11T16:19:40.478194Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def detect_outliers_iqr(df):\n    outlier_indices = []\n\n    for column in df.select_dtypes(include=['float64', 'int64']).columns:\n        Q1 = df[column].quantile(0.25)\n        Q3 = df[column].quantile(0.75)\n        IQR = Q3 - Q1\n\n        lower_bound = Q1 - 1.5 * IQR\n        upper_bound = Q3 + 1.5 * IQR\n\n        outliers = df[(df[column] < lower_bound) | (df[column] > upper_bound)]\n        outlier_indices.extend(outliers.index)\n\n        print(f'Outliers in {column}:', outliers.shape[0])\n\n    return outlier_indices","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:40.625961Z","iopub.execute_input":"2024-10-11T16:19:40.626482Z","iopub.status.idle":"2024-10-11T16:19:40.638987Z","shell.execute_reply.started":"2024-10-11T16:19:40.626435Z","shell.execute_reply":"2024-10-11T16:19:40.637359Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"outlier_indices = detect_outliers_iqr(data)\nprint(\"Total outliers detected:\", len(set(outlier_indices)))","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:40.888585Z","iopub.execute_input":"2024-10-11T16:19:40.889016Z","iopub.status.idle":"2024-10-11T16:19:41.051186Z","shell.execute_reply.started":"2024-10-11T16:19:40.888975Z","shell.execute_reply":"2024-10-11T16:19:41.050090Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Analysis","metadata":{}},{"cell_type":"code","source":"data.hist(figsize=(70, 40), bins=30)  \nplt.tight_layout()  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:19:41.152511Z","iopub.execute_input":"2024-10-11T16:19:41.152978Z","iopub.status.idle":"2024-10-11T16:20:03.651888Z","shell.execute_reply.started":"2024-10-11T16:19:41.152935Z","shell.execute_reply":"2024-10-11T16:20:03.650746Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## How internet usage affect physical measures ","metadata":{}},{"cell_type":"markdown","source":"### Average BMI , HeartRate , Blood pressure","metadata":{}},{"cell_type":"code","source":"averages = {\n    'BMI': data['Physical-BMI'].mean(),\n    'Heart Rate': data['Physical-HeartRate'].mean(),\n    'Systolic BP': data['Physical-Systolic_BP'].mean(),\n    'Diastolic BP': data['Physical-Diastolic_BP'].mean()\n}\n\n# Convert to DataFrame for easier plotting\naverages_df = pd.DataFrame(list(averages.items()), columns=['Feature', 'Average'])\n\n# Step 2: Create a bar plot\nplt.figure(figsize=(10, 6))\nsns.barplot(x='Feature', y='Average', data=averages_df, palette='coolwarm')\nplt.title('Average Values of BMI, Heart Rate, and Blood Pressure')\nplt.ylabel('Average')\nplt.xlabel('Features')\nplt.ylim(0, averages_df['Average'].max() + 10)  # Adjusting y-axis for better visibility\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:03.654090Z","iopub.execute_input":"2024-10-11T16:20:03.654505Z","iopub.status.idle":"2024-10-11T16:20:03.986099Z","shell.execute_reply.started":"2024-10-11T16:20:03.654461Z","shell.execute_reply":"2024-10-11T16:20:03.984755Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Childern Global Assessment Scale vs internet usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.histplot(data['CGAS-CGAS_Score'], bins=30, kde=True)  \nplt.title('Distribution of Childrens Global Assessment Scale Score')\nplt.xlabel('Assessment score')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:03.987450Z","iopub.execute_input":"2024-10-11T16:20:03.987808Z","iopub.status.idle":"2024-10-11T16:20:04.560801Z","shell.execute_reply.started":"2024-10-11T16:20:03.987770Z","shell.execute_reply":"2024-10-11T16:20:04.559564Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Assuming 'data' is your DataFrame\nmean_values = data.groupby('sii')['CGAS-CGAS_Score'].mean().reset_index()\n\n# Create a bar plot\nplt.figure(figsize=(10, 8))\nsns.barplot(x='sii', y='CGAS-CGAS_Score', data=mean_values, palette='viridis')\n\n# Customize the plot\nplt.title('Average Fitness Endurance Max Stage vs Total Internet Usage')\nplt.xlabel('Total Internet Usage (hours/day)')\nplt.ylabel('Average Fitness Endurance Max Stage')\nplt.ylim(0, 100)  # Set y-axis limits from 0 to 100\nplt.grid(axis='y')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:04.564125Z","iopub.execute_input":"2024-10-11T16:20:04.564651Z","iopub.status.idle":"2024-10-11T16:20:04.895262Z","shell.execute_reply.started":"2024-10-11T16:20:04.564592Z","shell.execute_reply":"2024-10-11T16:20:04.894133Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### BMI vs SII Severity","metadata":{}},{"cell_type":"code","source":"# Plot 1: BMI vs SII Severity\nplt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-BMI', palette='pastel')\nplt.title('BMI vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('BMI')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:04.896672Z","iopub.execute_input":"2024-10-11T16:20:04.897029Z","iopub.status.idle":"2024-10-11T16:20:05.228947Z","shell.execute_reply.started":"2024-10-11T16:20:04.896990Z","shell.execute_reply":"2024-10-11T16:20:05.227617Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Blood pressure vs internet Usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-Systolic_BP', palette='pastel')\nplt.title('Blood Pressure vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('Systolic BP')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:05.230663Z","iopub.execute_input":"2024-10-11T16:20:05.231132Z","iopub.status.idle":"2024-10-11T16:20:05.587118Z","shell.execute_reply.started":"2024-10-11T16:20:05.231077Z","shell.execute_reply":"2024-10-11T16:20:05.585888Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Heart rate vs internet usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-HeartRate', palette='pastel')\nplt.title('Heart Rate vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('Heart Rate')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:05.588537Z","iopub.execute_input":"2024-10-11T16:20:05.589003Z","iopub.status.idle":"2024-10-11T16:20:05.929511Z","shell.execute_reply.started":"2024-10-11T16:20:05.588948Z","shell.execute_reply":"2024-10-11T16:20:05.928217Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Demographics influence  ","metadata":{}},{"cell_type":"markdown","source":"### Age Distribution ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.histplot(data['Basic_Demos-Age'], bins=30, kde=True)  \nplt.title('Distribution of Age Feature')\nplt.xlabel('AGE')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:05.931131Z","iopub.execute_input":"2024-10-11T16:20:05.931713Z","iopub.status.idle":"2024-10-11T16:20:06.529666Z","shell.execute_reply.started":"2024-10-11T16:20:05.931664Z","shell.execute_reply":"2024-10-11T16:20:06.528385Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Gender Distribution","metadata":{}},{"cell_type":"code","source":"# Count the number of occurrences for each sex\nGende_counts = data['Basic_Demos-Sex'].value_counts()\n\n# Create the pie chart\nplt.figure(figsize=(7, 7))\nplt.pie(Gende_counts, labels=Gende_counts.index, autopct='%1.1f%%', startangle=90, colors=sns.color_palette('pastel'))\nplt.title('Distribution of Sex in the Dataset')\nplt.axis('equal')  # Equal aspect ratio ensures that pie chart is circular\nplt.show()\n# 0-> male  1_> female","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:06.531338Z","iopub.execute_input":"2024-10-11T16:20:06.531823Z","iopub.status.idle":"2024-10-11T16:20:06.753516Z","shell.execute_reply.started":"2024-10-11T16:20:06.531766Z","shell.execute_reply":"2024-10-11T16:20:06.752106Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_severity(score):\n    if score <= 30:\n        return 'None'\n    elif score <= 49:\n        return 'Mild'\n    elif score <= 79:\n        return 'Moderate'\n    else:\n        return 'Severe'\n\ndata['Severity'] = data['PCIAT-PCIAT_Total'].apply(categorize_severity)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:06.758096Z","iopub.execute_input":"2024-10-11T16:20:06.758498Z","iopub.status.idle":"2024-10-11T16:20:06.769892Z","shell.execute_reply.started":"2024-10-11T16:20:06.758455Z","shell.execute_reply":"2024-10-11T16:20:06.768465Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(12, 6))\nsns.boxplot(x='Severity', y='Basic_Demos-Age', data=data, palette='pastel')\nplt.title('Age Distribution Across Severity Impairment Index')\nplt.xlabel('Severity Impairment Index')\nplt.ylabel('Age')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:06.771750Z","iopub.execute_input":"2024-10-11T16:20:06.772227Z","iopub.status.idle":"2024-10-11T16:20:07.180664Z","shell.execute_reply.started":"2024-10-11T16:20:06.772174Z","shell.execute_reply":"2024-10-11T16:20:07.179348Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Calculate the average age for each severity category\naverage_age = data.groupby('Severity')['Basic_Demos-Age'].mean().reset_index()\n\n# Create a line plot\nplt.figure(figsize=(12, 6))\nsns.lineplot(x='Severity', y='Basic_Demos-Age', data=average_age, marker='o')\nplt.title('Average Age Across Severity Impairment Index')\nplt.xlabel('Severity Impairment Index')\nplt.ylabel('Average Age')\nplt.xticks(rotation=45)  # Rotate x labels for better visibility\nplt.grid(True)  # Add grid for better readability\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:07.182066Z","iopub.execute_input":"2024-10-11T16:20:07.182473Z","iopub.status.idle":"2024-10-11T16:20:07.559096Z","shell.execute_reply.started":"2024-10-11T16:20:07.182428Z","shell.execute_reply":"2024-10-11T16:20:07.557857Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ninternet_usage = data.groupby(['Basic_Demos-Age', 'PreInt_EduHx-computerinternet_hoursday']).size().unstack(fill_value=0)\n\ninternet_usage = internet_usage.reset_index()\n\ninternet_usage_melted = internet_usage.melt(id_vars='Basic_Demos-Age', \n                                              var_name='Internet Usage', \n                                              value_name='Count')\nplt.figure(figsize=(12, 6))\npalette = sns.color_palette('pastel') \nsns.barplot(data=internet_usage_melted, \n            x='Basic_Demos-Age', \n            y='Count', \n            hue='Internet Usage', \n            palette=palette)\n\nplt.xlabel('Age Group', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.title('Internet Usage Distribution by Age Group', fontsize=14)\nplt.xticks(rotation=45)\n\nhandles = []\nfor i, label in enumerate(['0=Less than 1h/day', '1=Around 1h/day', '2=Around 2hs/day', '3=More than 3hs/day']):\n    handles.append(plt.Line2D([0], [0], color=palette[i], lw=4))  # Create a line for each color\n\nplt.legend(handles, \n           ['0=Less than 1h/day', '1=Around 1h/day', '2=Around 2hs/day', '3=More than 3hs/day'], \n           title='Daily Internet Usage')\n\nplt.tight_layout() \nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:07.560540Z","iopub.execute_input":"2024-10-11T16:20:07.560890Z","iopub.status.idle":"2024-10-11T16:20:08.509533Z","shell.execute_reply.started":"2024-10-11T16:20:07.560851Z","shell.execute_reply":"2024-10-11T16:20:08.508508Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sleep Quality vs internet usage","metadata":{}},{"cell_type":"code","source":"mean_sleep_time = data.groupby('PreInt_EduHx-computerinternet_hoursday')['SDS-SDS_Total_T'].mean()\n\nplt.figure(figsize=(10, 8))\nplt.plot(mean_sleep_time.index, mean_sleep_time.values, marker='o', linestyle='-', color='blue')\n\n# Customize the plot\nplt.xlabel('Internet Usage Time (hours/day)')\nplt.ylabel('Average Sleep Time (hours)')\nplt.title('Average Sleep Time vs Internet Usage Time')\n\ncustom_ticks = [ 45, 50, 55, 60, 65, 70,75]  # Define your custom tick values\nplt.yticks(custom_ticks)\n\nplt.grid(True)\n\n# Show the plot\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:08.511028Z","iopub.execute_input":"2024-10-11T16:20:08.511553Z","iopub.status.idle":"2024-10-11T16:20:08.871981Z","shell.execute_reply.started":"2024-10-11T16:20:08.511497Z","shell.execute_reply":"2024-10-11T16:20:08.870794Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Behavioural influence ","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n\nplt.figure(figsize=(12, 6))\nsns.boxplot(x='PCIAT-PCIAT_06', y='Basic_Demos-Age', data=data, palette='pastel')\nplt.title('Age Distribution for Q: How often do your childs grades suffer because of the amount of time he or she spends online?')\nplt.xlabel('Response')\nplt.ylabel('Age')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:08.873812Z","iopub.execute_input":"2024-10-11T16:20:08.874322Z","iopub.status.idle":"2024-10-11T16:20:09.298302Z","shell.execute_reply.started":"2024-10-11T16:20:08.874242Z","shell.execute_reply":"2024-10-11T16:20:09.296987Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# List of questions to analyze\nquestions = [\n    'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', \n    'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06',\n    'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09',\n    'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12',\n    'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15',\n    'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18',\n    'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20'\n]\n\nplt.figure(figsize=(15, 40))  # Adjust the size for better visibility\n\n# Create a box plot for each question\nfor i, question in enumerate(questions):\n    plt.subplot(len(questions), 1, i + 1)  # Create subplots\n    sns.boxplot(x=question, y='Basic_Demos-Age', data=data, palette='pastel')\n    plt.title(f'Age Distribution for {question}')\n    plt.xlabel('Response')\n    plt.ylabel('Age')\n\nplt.tight_layout()  # Adjust layout to prevent overlap\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:09.300109Z","iopub.execute_input":"2024-10-11T16:20:09.300653Z","iopub.status.idle":"2024-10-11T16:20:15.588370Z","shell.execute_reply.started":"2024-10-11T16:20:09.300596Z","shell.execute_reply":"2024-10-11T16:20:15.587145Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"markdown","source":"## Handling Missing Values ","metadata":{}},{"cell_type":"markdown","source":"### Drop columns with null>50","metadata":{}},{"cell_type":"code","source":"\ncol = ['CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n       'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season',\n       'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']\ncol.extend( \n        ['Physical-Waist_Circumference','Fitness_Endurance-Max_Stage',\n         'Fitness_Endurance-Time_Mins','Fitness_Endurance-Time_Sec','FGC-FGC_GSND',\n         'FGC-FGC_GSND_Zone','FGC-FGC_GSD','FGC-FGC_GSD_Zone','BIA-BIA_Activity_Level_num',\n         'BIA-BIA_BMC','BIA-BIA_BMI','BIA-BIA_BMR','BIA-BIA_DEE','BIA-BIA_ECW','BIA-BIA_FFM',\n         'BIA-BIA_FFMI','BIA-BIA_Fat','BIA-BIA_Frame_num','BIA-BIA_ICW','BIA-BIA_LDM','BIA-BIA_LST',\n         'BIA-BIA_SMM','BIA-BIA_TBW','PAQ_A-PAQ_A_Total','PAQ_C-PAQ_C_Total'])\nlen(col)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:15.589907Z","iopub.execute_input":"2024-10-11T16:20:15.590303Z","iopub.status.idle":"2024-10-11T16:20:15.600941Z","shell.execute_reply.started":"2024-10-11T16:20:15.590240Z","shell.execute_reply":"2024-10-11T16:20:15.599632Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.drop(col , axis =1 ,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:15.602397Z","iopub.execute_input":"2024-10-11T16:20:15.602774Z","iopub.status.idle":"2024-10-11T16:20:15.618229Z","shell.execute_reply.started":"2024-10-11T16:20:15.602726Z","shell.execute_reply":"2024-10-11T16:20:15.616739Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:15.619940Z","iopub.execute_input":"2024-10-11T16:20:15.620462Z","iopub.status.idle":"2024-10-11T16:20:15.631728Z","shell.execute_reply.started":"2024-10-11T16:20:15.620398Z","shell.execute_reply":"2024-10-11T16:20:15.630393Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned= data.copy()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:15.633354Z","iopub.execute_input":"2024-10-11T16:20:15.633724Z","iopub.status.idle":"2024-10-11T16:20:15.643567Z","shell.execute_reply.started":"2024-10-11T16:20:15.633685Z","shell.execute_reply":"2024-10-11T16:20:15.642307Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:15.645016Z","iopub.execute_input":"2024-10-11T16:20:15.645448Z","iopub.status.idle":"2024-10-11T16:20:15.656910Z","shell.execute_reply.started":"2024-10-11T16:20:15.645405Z","shell.execute_reply":"2024-10-11T16:20:15.655759Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Filling Missing Values in first 46 columns with Knn imputer","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\ndata_cleaned['Basic_Demos-Enroll_Season'] = label_encoder.fit_transform(data_cleaned['Basic_Demos-Enroll_Season'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:15.658517Z","iopub.execute_input":"2024-10-11T16:20:15.658892Z","iopub.status.idle":"2024-10-11T16:20:15.669656Z","shell.execute_reply.started":"2024-10-11T16:20:15.658853Z","shell.execute_reply":"2024-10-11T16:20:15.668519Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\nimputer = KNNImputer(n_neighbors=5)\ndata_cleaned.iloc[:, :45] = imputer.fit_transform(data_cleaned.iloc[:, :45])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:15.671254Z","iopub.execute_input":"2024-10-11T16:20:15.671731Z","iopub.status.idle":"2024-10-11T16:20:20.483458Z","shell.execute_reply.started":"2024-10-11T16:20:15.671690Z","shell.execute_reply":"2024-10-11T16:20:20.482368Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\ndata_cleaned['Severity'] = label_encoder.fit_transform(data_cleaned['Severity'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:20.484881Z","iopub.execute_input":"2024-10-11T16:20:20.485247Z","iopub.status.idle":"2024-10-11T16:20:20.493036Z","shell.execute_reply.started":"2024-10-11T16:20:20.485207Z","shell.execute_reply":"2024-10-11T16:20:20.491768Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Filling missing values in target with Kmeans","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nimport numpy as np\nimport pandas as pd\n\ndef impute_with_kmeans(df, categorical_columns, n_clusters=4):\n    # Fill missing values temporarily with the mode (or any placeholder)\n    df_temp = df.copy()\n    for col in categorical_columns:\n        df_temp[col].fillna(df_temp[col].mode()[0], inplace=True)\n\n    # Perform KMeans clustering after filling missing values\n    kmeans = KMeans(n_clusters=n_clusters, random_state=0)\n    cluster_labels = kmeans.fit_predict(df_temp)\n\n    # Impute missing values within each cluster\n    for col in categorical_columns:\n        for cluster in np.unique(cluster_labels):\n            mask = (cluster_labels == cluster) & df[col].isna()\n            most_frequent = df.loc[cluster_labels == cluster, col].mode()[0]\n            df.loc[mask, col] = most_frequent\n\n    return df\n\n# Apply KMeans-based imputation\ndata_cleaned = impute_with_kmeans(data_cleaned, ['sii'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:20.494656Z","iopub.execute_input":"2024-10-11T16:20:20.495037Z","iopub.status.idle":"2024-10-11T16:20:21.298682Z","shell.execute_reply.started":"2024-10-11T16:20:20.494995Z","shell.execute_reply":"2024-10-11T16:20:21.297393Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned['sii'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:21.300189Z","iopub.execute_input":"2024-10-11T16:20:21.300575Z","iopub.status.idle":"2024-10-11T16:20:21.312695Z","shell.execute_reply.started":"2024-10-11T16:20:21.300532Z","shell.execute_reply":"2024-10-11T16:20:21.311340Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:21.314624Z","iopub.execute_input":"2024-10-11T16:20:21.315039Z","iopub.status.idle":"2024-10-11T16:20:21.348760Z","shell.execute_reply.started":"2024-10-11T16:20:21.314997Z","shell.execute_reply":"2024-10-11T16:20:21.347532Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_cleaned.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:21.356966Z","iopub.execute_input":"2024-10-11T16:20:21.357428Z","iopub.status.idle":"2024-10-11T16:20:21.368718Z","shell.execute_reply.started":"2024-10-11T16:20:21.357384Z","shell.execute_reply":"2024-10-11T16:20:21.367476Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handling outliers ","metadata":{}},{"cell_type":"code","source":"outlier_indices = detect_outliers_iqr(data_cleaned)\nprint(\"Total outliers detected:\", len(set(outlier_indices)))","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:21.370162Z","iopub.execute_input":"2024-10-11T16:20:21.370575Z","iopub.status.idle":"2024-10-11T16:20:21.497091Z","shell.execute_reply.started":"2024-10-11T16:20:21.370527Z","shell.execute_reply":"2024-10-11T16:20:21.496043Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for column in data_cleaned.columns:\n    plt.figure(figsize=(8, 3))  \n    sns.boxplot(data=data_cleaned[[column]], orient='h', palette=\"Set2\")\n    \n    plt.title(f'Box Plot for {column} with Outliers', fontsize=16)\n    plt.ylabel(column, fontsize=14)\n    plt.xlabel('Values', fontsize=14)\n    \n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:21.498377Z","iopub.execute_input":"2024-10-11T16:20:21.498730Z","iopub.status.idle":"2024-10-11T16:20:38.534417Z","shell.execute_reply.started":"2024-10-11T16:20:21.498692Z","shell.execute_reply":"2024-10-11T16:20:38.533302Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate IQR for the column 'CGAS-CGAS_Score'\nQ1 = data_cleaned['CGAS-CGAS_Score'].quantile(0.25)\nQ3 = data_cleaned['CGAS-CGAS_Score'].quantile(0.75)\nIQR = Q3 - Q1\n\n# Define outlier boundaries\nlower_bound = Q1 - 1.5 * IQR\nupper_bound = Q3 + 1.5 * IQR\n\n# Filter the entire DataFrame by keeping only rows where 'CGAS-CGAS_Score' is within bounds\ndata_copy = data_cleaned[(data_cleaned['CGAS-CGAS_Score'] >= lower_bound) & (data_cleaned['CGAS-CGAS_Score'] <= upper_bound)]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:38.535932Z","iopub.execute_input":"2024-10-11T16:20:38.536324Z","iopub.status.idle":"2024-10-11T16:20:38.548260Z","shell.execute_reply.started":"2024-10-11T16:20:38.536253Z","shell.execute_reply":"2024-10-11T16:20:38.547130Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"## BMI To Heart rate Ratio","metadata":{}},{"cell_type":"code","source":"data_copy['HeartRate_BMI'] = data_copy['Physical-HeartRate'] * data_copy['Physical-BMI']","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:38.549926Z","iopub.execute_input":"2024-10-11T16:20:38.550526Z","iopub.status.idle":"2024-10-11T16:20:38.559747Z","shell.execute_reply.started":"2024-10-11T16:20:38.550479Z","shell.execute_reply":"2024-10-11T16:20:38.558536Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Max Heart rate due to age","metadata":{}},{"cell_type":"code","source":"data_copy['HRmax'] = 220 - data_copy['Basic_Demos-Age']  # Estimate HRmax based on age","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:38.563893Z","iopub.execute_input":"2024-10-11T16:20:38.565126Z","iopub.status.idle":"2024-10-11T16:20:38.575924Z","shell.execute_reply.started":"2024-10-11T16:20:38.565076Z","shell.execute_reply":"2024-10-11T16:20:38.574637Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Addation Feature ","metadata":{}},{"cell_type":"code","source":"def categorize_bmi(bmi):\n    if bmi <= 18.5:\n        return 0\n    elif bmi <= 24.9:\n        return 1\n    elif bmi <= 29.9:\n        return 2\n    elif bmi <= 40:\n        return 3\n    else:\n        return 4  \n\ndata_copy['BMI_Category'] = data_copy['Physical-BMI'].apply(categorize_bmi)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:20:38.577525Z","iopub.execute_input":"2024-10-11T16:20:38.577921Z","iopub.status.idle":"2024-10-11T16:20:38.591913Z","shell.execute_reply.started":"2024-10-11T16:20:38.577879Z","shell.execute_reply":"2024-10-11T16:20:38.590735Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pciat_columns = [f'PCIAT-PCIAT_{i:02d}' for i in range(1, 21)]\ndata_copy['PCIAT_Time_Management'] = data_copy[pciat_columns[:5]].mean(axis=1)\ndata_copy['PCIAT_Withdrawal_Symptoms'] = data_copy[pciat_columns[5:10]].mean(axis=1)\ndata_copy['PCIAT_Neglect_Social_Life'] = data_copy[pciat_columns[10:15]].mean(axis=1)\ndata_copy['PCIAT_Lack_Control'] = data_copy[pciat_columns[15:]].mean(axis=1)\n\n\n\ndata_copy['PCIAT_mean'] = data_copy[pciat_columns].mean(axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:20:38.593400Z","iopub.execute_input":"2024-10-11T16:20:38.593781Z","iopub.status.idle":"2024-10-11T16:20:38.617772Z","shell.execute_reply.started":"2024-10-11T16:20:38.593740Z","shell.execute_reply":"2024-10-11T16:20:38.616589Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_pciat(score):\n    if score <= 20:\n         return 0\n    elif score <= 49:\n          return 1\n    elif score <= 79:\n         return 2\n    else:\n         return 3\n    \ndata_copy['PCIAT_Category'] = data_copy['PCIAT-PCIAT_Total'].apply(categorize_pciat)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:20:38.618989Z","iopub.execute_input":"2024-10-11T16:20:38.619338Z","iopub.status.idle":"2024-10-11T16:20:38.630572Z","shell.execute_reply.started":"2024-10-11T16:20:38.619294Z","shell.execute_reply":"2024-10-11T16:20:38.629343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def categorize_sds(score):\n    if pd.isna(score) or score < 0:\n        return np.nan\n    elif score <= 20:\n        return 0\n    elif score <= 40:\n        return 1\n    elif score <= 60:\n        return 2\n    elif score <= 80:\n        return 3\n    else:\n        return 4  \n\ndata_copy['SDS_Severity'] = data_copy['SDS-SDS_Total_Raw'].apply(categorize_sds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:20:38.632265Z","iopub.execute_input":"2024-10-11T16:20:38.632757Z","iopub.status.idle":"2024-10-11T16:20:38.651653Z","shell.execute_reply.started":"2024-10-11T16:20:38.632705Z","shell.execute_reply":"2024-10-11T16:20:38.650503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['Sleep_Quality_Index'] = (data_copy['SDS-SDS_Total_T'] - data_copy['SDS-SDS_Total_T'].min()) / (data_copy['SDS-SDS_Total_T'].max() - data_copy['SDS-SDS_Total_T'].min())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:20:38.653069Z","iopub.execute_input":"2024-10-11T16:20:38.655896Z","iopub.status.idle":"2024-10-11T16:20:38.666545Z","shell.execute_reply.started":"2024-10-11T16:20:38.655850Z","shell.execute_reply":"2024-10-11T16:20:38.665402Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['Physical_Health_Index'] = ((data_copy['Physical-BMI'] - data_copy['Physical-BMI'].mean()) / data_copy['Physical-BMI'].std() + (data_copy['Physical-Systolic_BP'] - data_copy['Physical-Systolic_BP'].mean()) / data_copy['Physical-Systolic_BP'].std() + (data_copy['Physical-HeartRate'] - data_copy['Physical-HeartRate'].mean()) / data_copy['Physical-HeartRate'].std()) / 3\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:20:38.668389Z","iopub.execute_input":"2024-10-11T16:20:38.668862Z","iopub.status.idle":"2024-10-11T16:20:38.683074Z","shell.execute_reply.started":"2024-10-11T16:20:38.668821Z","shell.execute_reply":"2024-10-11T16:20:38.681701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy['Sleep_Quality_Index'] = (data_copy['SDS-SDS_Total_T'] - data_copy['SDS-SDS_Total_T'].min()) / (data_copy['SDS-SDS_Total_T'].max() - data_copy['SDS-SDS_Total_T'].min())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:20:38.684675Z","iopub.execute_input":"2024-10-11T16:20:38.685088Z","iopub.status.idle":"2024-10-11T16:20:38.697843Z","shell.execute_reply.started":"2024-10-11T16:20:38.685046Z","shell.execute_reply":"2024-10-11T16:20:38.696436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fgc_columns = ['FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_TL']\n\ndata_copy['Overall_Fitness_Score'] = data_copy[fgc_columns].mean(axis=1)\ndata_copy['Internet_Usage_Score'] = data_copy['PCIAT-PCIAT_Total'] / 100\ndata_copy['Physical_Activity_Score'] = data_copy['Overall_Fitness_Score'] / data_copy['Overall_Fitness_Score'].max()\ndata_copy['Lifestyle_Score'] = ((1 - data_copy['Internet_Usage_Score']) +  data_copy['Physical_Activity_Score'] + (1 - data_copy['Sleep_Quality_Index'])) / 3","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:20:38.699500Z","iopub.execute_input":"2024-10-11T16:20:38.700088Z","iopub.status.idle":"2024-10-11T16:20:38.715976Z","shell.execute_reply.started":"2024-10-11T16:20:38.700045Z","shell.execute_reply":"2024-10-11T16:20:38.714871Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy = data_copy.drop(columns=['SDS-SDS_Total_T', 'SDS-SDS_Total_Raw', 'PCIAT-PCIAT_Total', 'Physical-Height', 'Physical-Weight', 'FGC-FGC_CU', 'FGC-FGC_PU', 'FGC-FGC_SRL', 'FGC-FGC_SRR', 'FGC-FGC_TL', 'FGC-FGC_CU_Zone', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRR_Zone', 'FGC-FGC_SRL_Zone', 'FGC-FGC_TL_Zone', 'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', 'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06', 'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09', 'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12', 'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15', 'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18', 'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20'])\n\n\n\ndata_copy = data_copy.drop(columns=['PCIAT_Time_Management', 'PCIAT_Withdrawal_Symptoms', 'PCIAT_Neglect_Social_Life', 'PCIAT_Lack_Control',  'Basic_Demos-Age', 'Physical-BMI'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:20:38.717130Z","iopub.execute_input":"2024-10-11T16:20:38.717522Z","iopub.status.idle":"2024-10-11T16:20:38.732555Z","shell.execute_reply.started":"2024-10-11T16:20:38.717481Z","shell.execute_reply":"2024-10-11T16:20:38.731308Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Body Strength & Flexibility","metadata":{}},{"cell_type":"code","source":"# data_copy['BodyStrength&Flexibility'] = data_copy['FGC-FGC_CU'] + data_copy['FGC-FGC_PU'] + data_copy['FGC-FGC_SRL'] + data_copy['FGC-FGC_SRR'] + data_copy['FGC-FGC_TL']\n# data_copy['BodyStrength&Flexibility_class'] = data_copy['FGC-FGC_CU_Zone'] + data_copy['FGC-FGC_PU_Zone'] + data_copy['FGC-FGC_SRL_Zone'] + data_copy['FGC-FGC_SRR_Zone'] + data_copy['FGC-FGC_TL_Zone']\n\n# data_copy = data_copy.drop(['FGC-FGC_CU','FGC-FGC_PU','FGC-FGC_SRL','FGC-FGC_SRR','FGC-FGC_TL'], axis=1)\n\n# data_copy = data_copy.drop(['FGC-FGC_CU_Zone','FGC-FGC_PU_Zone','FGC-FGC_SRL_Zone','FGC-FGC_SRR_Zone','FGC-FGC_TL_Zone'], axis=1)                          \n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:38.734144Z","iopub.execute_input":"2024-10-11T16:20:38.734546Z","iopub.status.idle":"2024-10-11T16:20:38.740676Z","shell.execute_reply.started":"2024-10-11T16:20:38.734504Z","shell.execute_reply":"2024-10-11T16:20:38.739494Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_copy.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:38.742107Z","iopub.execute_input":"2024-10-11T16:20:38.742618Z","iopub.status.idle":"2024-10-11T16:20:38.758720Z","shell.execute_reply.started":"2024-10-11T16:20:38.742565Z","shell.execute_reply":"2024-10-11T16:20:38.757419Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Transfomation and scaling","metadata":{}},{"cell_type":"markdown","source":"## Data Splitting ","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = data_copy.drop('sii', axis=1)  \ny = data_copy['sii']  \nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)\nprint(X_train.shape, X_test.shape, y_train.shape, y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:38.760533Z","iopub.execute_input":"2024-10-11T16:20:38.760941Z","iopub.status.idle":"2024-10-11T16:20:38.776484Z","shell.execute_reply.started":"2024-10-11T16:20:38.760900Z","shell.execute_reply":"2024-10-11T16:20:38.775316Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Handling imbalanced data ","metadata":{}},{"cell_type":"code","source":"from collections import Counter\nfrom imblearn.over_sampling import SMOTE\n\n# Check the class distribution before applying SMOTE\nclass_distribution_before = Counter(y_train)\nprint(\"Class distribution before SMOTE:\", class_distribution_before)\n\n# Initialize SMOTE with specified parameters\nsmote = SMOTE(sampling_strategy='auto', random_state=42)\n\n# Apply SMOTE to the training data\nX_train_resampled, y_train_resampled = smote.fit_resample(X_train, y_train)\n\n# Check the class distribution after applying SMOTE\nclass_distribution_after = Counter(y_train_resampled)\nprint(\"Class distribution after SMOTE:\", class_distribution_after)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:38.778190Z","iopub.execute_input":"2024-10-11T16:20:38.778965Z","iopub.status.idle":"2024-10-11T16:20:38.812074Z","shell.execute_reply.started":"2024-10-11T16:20:38.778912Z","shell.execute_reply":"2024-10-11T16:20:38.810923Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_train = y_train_resampled","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:20:38.813617Z","iopub.execute_input":"2024-10-11T16:20:38.813964Z","iopub.status.idle":"2024-10-11T16:20:38.819326Z","shell.execute_reply.started":"2024-10-11T16:20:38.813925Z","shell.execute_reply":"2024-10-11T16:20:38.818152Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Scaling","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import RobustScaler\nscaler = RobustScaler()  # or StandardScaler()\n\n# Fit the scaler on X_train and transform both X_train and X_test\nX_train = pd.DataFrame(scaler.fit_transform(X_train_resampled), columns=X_train_resampled.columns)\nX_test = pd.DataFrame(scaler.transform(X_test), columns=X_test.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:38.820882Z","iopub.execute_input":"2024-10-11T16:20:38.821320Z","iopub.status.idle":"2024-10-11T16:20:38.850736Z","shell.execute_reply.started":"2024-10-11T16:20:38.821239Z","shell.execute_reply":"2024-10-11T16:20:38.849592Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.shape\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:38.852043Z","iopub.execute_input":"2024-10-11T16:20:38.852408Z","iopub.status.idle":"2024-10-11T16:20:38.863823Z","shell.execute_reply.started":"2024-10-11T16:20:38.852369Z","shell.execute_reply":"2024-10-11T16:20:38.862653Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:20:38.865203Z","iopub.execute_input":"2024-10-11T16:20:38.865591Z","iopub.status.idle":"2024-10-11T16:20:38.872819Z","shell.execute_reply.started":"2024-10-11T16:20:38.865536Z","shell.execute_reply":"2024-10-11T16:20:38.871722Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PCA","metadata":{}},{"cell_type":"code","source":"from sklearn.decomposition import PCA \npca = PCA(n_components= 20 )  # Reduce to 2 components\nX_train = pca.fit_transform(X_train)\nX_test = pca.transform(X_test)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:20:38.874242Z","iopub.execute_input":"2024-10-11T16:20:38.874665Z","iopub.status.idle":"2024-10-11T16:20:39.666469Z","shell.execute_reply.started":"2024-10-11T16:20:38.874624Z","shell.execute_reply":"2024-10-11T16:20:39.664002Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Modeling ","metadata":{}},{"cell_type":"code","source":"pip install lazypredict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:20:39.671475Z","iopub.execute_input":"2024-10-11T16:20:39.671856Z","iopub.status.idle":"2024-10-11T16:20:53.730426Z","shell.execute_reply.started":"2024-10-11T16:20:39.671814Z","shell.execute_reply":"2024-10-11T16:20:53.728826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lazypredict.Supervised import LazyClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.datasets import load_iris\nimport pandas as pd\n\n\n\nclf = LazyClassifier(verbose=0, ignore_warnings=True, custom_metric=None)\n\nmodels, predictions = clf.fit(X_train, X_test, y_train, y_test)\n\nprint(models)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-10-11T16:20:53.732702Z","iopub.execute_input":"2024-10-11T16:20:53.733257Z","iopub.status.idle":"2024-10-11T16:21:24.853485Z","shell.execute_reply.started":"2024-10-11T16:20:53.733191Z","shell.execute_reply":"2024-10-11T16:21:24.852380Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result1, result2, result3 = [], [], [] ","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:21:24.855018Z","iopub.execute_input":"2024-10-11T16:21:24.855427Z","iopub.status.idle":"2024-10-11T16:21:24.862928Z","shell.execute_reply.started":"2024-10-11T16:21:24.855385Z","shell.execute_reply":"2024-10-11T16:21:24.861763Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def modeling(model):\n    # Fit the model using the scaled resampled training data\n    model.fit(X_train, y_train_resampled)  # Use y_train_resampled here\n    \n    # Predictions\n    train_pred = model.predict(X_train)    # Predictions on the scaled resampled train set\n    test_pred = model.predict(X_test)      # Predictions on the scaled original test set\n    \n    # Calculate metrics with specified average\n    train_accuracy = accuracy_score(y_train_resampled, train_pred) * 100\n    train_recall = recall_score(y_train_resampled, train_pred, average='weighted') * 100\n    train_f1_score = f1_score(y_train_resampled, train_pred, average='weighted') * 100\n    \n    test_accuracy = accuracy_score(y_test, test_pred) * 100\n    test_recall = recall_score(y_test, test_pred, average='weighted') * 100\n    test_f1_score = f1_score(y_test, test_pred, average='weighted') * 100\n    \n    # Append results\n    result1.append(test_accuracy)\n    result2.append(test_recall)\n    result3.append(test_f1_score)\n    \n    print(\"Classification Report for Test Data:\")\n    print(classification_report(y_test, test_pred))\n    \n    print(\"\\nClassification Report for Scaled Resampled Train Data:\")\n    print(classification_report(y_train_resampled, train_pred))  # Use y_train_resampled here\n    \n    # Accuracy, Recall, and F1 Scores\n    print(f'Training Accuracy: {train_accuracy}, Train Recall: {train_recall}, Train F1: {train_f1_score}')\n    print(f'Test Accuracy: {test_accuracy}, Test Recall: {test_recall}, Test F1: {test_f1_score}')\n    \n    # Confusion matrix\n    cm = confusion_matrix(y_test, test_pred)\n    sns.heatmap(cm, annot=True, fmt='0.2f', cmap='YlGnBu', linewidths=1)\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n    plt.title('Confusion Matrix')\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:21:24.864472Z","iopub.execute_input":"2024-10-11T16:21:24.864841Z","iopub.status.idle":"2024-10-11T16:21:24.877968Z","shell.execute_reply.started":"2024-10-11T16:21:24.864801Z","shell.execute_reply":"2024-10-11T16:21:24.876857Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression  \nlogistic_regression = LogisticRegression()\nmodeling(logistic_regression)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:21:24.879804Z","iopub.execute_input":"2024-10-11T16:21:24.880895Z","iopub.status.idle":"2024-10-11T16:21:26.113469Z","shell.execute_reply.started":"2024-10-11T16:21:24.880832Z","shell.execute_reply":"2024-10-11T16:21:26.112230Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SVM = SVC()\nmodeling(SVM)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:21:26.115232Z","iopub.execute_input":"2024-10-11T16:21:26.115731Z","iopub.status.idle":"2024-10-11T16:21:27.511988Z","shell.execute_reply.started":"2024-10-11T16:21:26.115676Z","shell.execute_reply":"2024-10-11T16:21:27.510781Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decision_tree_classifier = DecisionTreeClassifier()\nmodeling(decision_tree_classifier)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:21:27.513967Z","iopub.execute_input":"2024-10-11T16:21:27.514478Z","iopub.status.idle":"2024-10-11T16:21:28.182309Z","shell.execute_reply.started":"2024-10-11T16:21:27.514422Z","shell.execute_reply":"2024-10-11T16:21:28.181087Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"decision_tree_classifier = DecisionTreeClassifier( max_depth= 7 , min_samples_split= 7)\nmodeling(decision_tree_classifier)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:21:28.184072Z","iopub.execute_input":"2024-10-11T16:21:28.184595Z","iopub.status.idle":"2024-10-11T16:21:28.753519Z","shell.execute_reply.started":"2024-10-11T16:21:28.184533Z","shell.execute_reply":"2024-10-11T16:21:28.752121Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sgd_classifier = SGDClassifier(max_iter = 500) \nmodeling(sgd_classifier)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:21:28.755368Z","iopub.execute_input":"2024-10-11T16:21:28.755874Z","iopub.status.idle":"2024-10-11T16:21:29.366756Z","shell.execute_reply.started":"2024-10-11T16:21:28.755820Z","shell.execute_reply":"2024-10-11T16:21:29.365567Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nknn = KNeighborsClassifier()\nmodeling(knn)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:21:29.368718Z","iopub.execute_input":"2024-10-11T16:21:29.369204Z","iopub.status.idle":"2024-10-11T16:21:30.582224Z","shell.execute_reply.started":"2024-10-11T16:21:29.369149Z","shell.execute_reply":"2024-10-11T16:21:30.581019Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n\n# Create an instance of GaussianNB\nnaive_bayes = GaussianNB()\n\n# Call your modeling function\nmodeling(naive_bayes)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:21:30.583722Z","iopub.execute_input":"2024-10-11T16:21:30.584089Z","iopub.status.idle":"2024-10-11T16:21:31.035338Z","shell.execute_reply.started":"2024-10-11T16:21:30.584048Z","shell.execute_reply":"2024-10-11T16:21:31.033839Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nrandom_forest = RandomForestClassifier(random_state=1)\nmodeling(random_forest)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:21:31.039713Z","iopub.execute_input":"2024-10-11T16:21:31.040157Z","iopub.status.idle":"2024-10-11T16:21:34.281669Z","shell.execute_reply.started":"2024-10-11T16:21:31.040088Z","shell.execute_reply":"2024-10-11T16:21:34.280353Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"adaBosster=AdaBoostClassifier( n_estimators=50)\nmodeling(adaBosster)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:21:34.283374Z","iopub.execute_input":"2024-10-11T16:21:34.283849Z","iopub.status.idle":"2024-10-11T16:21:36.099745Z","shell.execute_reply.started":"2024-10-11T16:21:34.283791Z","shell.execute_reply":"2024-10-11T16:21:36.098455Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingClassifier\n\ngradient=GradientBoostingClassifier()\nmodeling(gradient)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T16:21:36.101700Z","iopub.execute_input":"2024-10-11T16:21:36.102106Z","iopub.status.idle":"2024-10-11T16:21:57.699238Z","shell.execute_reply.started":"2024-10-11T16:21:36.102062Z","shell.execute_reply":"2024-10-11T16:21:57.697939Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}