{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt \nimport pandas as pd \nimport plotly.express as px\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport math\nfrom plotly.subplots import make_subplots\nimport numpy as np\nfrom numpy import linalg as LA\nimport plotly.express as px\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\nimport warnings\n\nwarnings.filterwarnings('ignore')\n\nfrom sklearn.ensemble import AdaBoostClassifier\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nfrom sklearn.svm import SVC\nimport numpy as np\nimport matplotlib.pyplot as plt \nimport pandas as pd \nimport plotly.express as px\nimport seaborn as sns\nimport plotly.graph_objects as go\nimport math\nfrom plotly.subplots import make_subplots\nimport numpy as np\nfrom numpy import linalg as LA\nimport plotly.express as px\nimport pandas as pd\nfrom sklearn.preprocessing import LabelEncoder\n\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import recall_score, f1_score, accuracy_score\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import (accuracy_score, recall_score, f1_score, \n                             classification_report, confusion_matrix)\nfrom sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:43.826045Z","iopub.execute_input":"2024-10-11T10:16:43.826515Z","iopub.status.idle":"2024-10-11T10:16:43.837937Z","shell.execute_reply.started":"2024-10-11T10:16:43.826469Z","shell.execute_reply":"2024-10-11T10:16:43.836713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Exploration ","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/child-mind-institute-problematic-internet-use/train.csv\")\npd.set_option('display.max_columns', None)\npd.set_option('display.max_rows', None)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:43.840014Z","iopub.execute_input":"2024-10-11T10:16:43.840419Z","iopub.status.idle":"2024-10-11T10:16:43.905779Z","shell.execute_reply.started":"2024-10-11T10:16:43.840377Z","shell.execute_reply":"2024-10-11T10:16:43.904661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:43.907811Z","iopub.execute_input":"2024-10-11T10:16:43.908197Z","iopub.status.idle":"2024-10-11T10:16:43.915651Z","shell.execute_reply.started":"2024-10-11T10:16:43.908154Z","shell.execute_reply":"2024-10-11T10:16:43.914453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:43.917333Z","iopub.execute_input":"2024-10-11T10:16:43.917769Z","iopub.status.idle":"2024-10-11T10:16:43.970566Z","shell.execute_reply.started":"2024-10-11T10:16:43.917715Z","shell.execute_reply":"2024-10-11T10:16:43.969313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:44.530316Z","iopub.execute_input":"2024-10-11T10:16:44.530814Z","iopub.status.idle":"2024-10-11T10:16:44.551637Z","shell.execute_reply.started":"2024-10-11T10:16:44.530768Z","shell.execute_reply":"2024-10-11T10:16:44.550493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop('id', axis=1 , inplace = True)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:44.553793Z","iopub.execute_input":"2024-10-11T10:16:44.554300Z","iopub.status.idle":"2024-10-11T10:16:44.566877Z","shell.execute_reply.started":"2024-10-11T10:16:44.554245Z","shell.execute_reply":"2024-10-11T10:16:44.565731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:44.568550Z","iopub.execute_input":"2024-10-11T10:16:44.569051Z","iopub.status.idle":"2024-10-11T10:16:44.604147Z","shell.execute_reply.started":"2024-10-11T10:16:44.568994Z","shell.execute_reply":"2024-10-11T10:16:44.603074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop_duplicates(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:44.606503Z","iopub.execute_input":"2024-10-11T10:16:44.606892Z","iopub.status.idle":"2024-10-11T10:16:44.635611Z","shell.execute_reply.started":"2024-10-11T10:16:44.606853Z","shell.execute_reply":"2024-10-11T10:16:44.634398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:44.636960Z","iopub.execute_input":"2024-10-11T10:16:44.637449Z","iopub.status.idle":"2024-10-11T10:16:44.668828Z","shell.execute_reply.started":"2024-10-11T10:16:44.637377Z","shell.execute_reply":"2024-10-11T10:16:44.667694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:44.670220Z","iopub.execute_input":"2024-10-11T10:16:44.670618Z","iopub.status.idle":"2024-10-11T10:16:44.854900Z","shell.execute_reply.started":"2024-10-11T10:16:44.670576Z","shell.execute_reply":"2024-10-11T10:16:44.853648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:44.856533Z","iopub.execute_input":"2024-10-11T10:16:44.856980Z","iopub.status.idle":"2024-10-11T10:16:44.871024Z","shell.execute_reply.started":"2024-10-11T10:16:44.856937Z","shell.execute_reply":"2024-10-11T10:16:44.869578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols = [col for col in data.columns if data[col].dtype != 'O']\ncat_cols = [col for col in data.columns if col not in num_cols]\nprint(f'Numerical columns: {num_cols}')\nprint(\"-------------------\")\nprint(f'Categorical columns: {cat_cols}')","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:44.872742Z","iopub.execute_input":"2024-10-11T10:16:44.873267Z","iopub.status.idle":"2024-10-11T10:16:44.883330Z","shell.execute_reply.started":"2024-10-11T10:16:44.873222Z","shell.execute_reply":"2024-10-11T10:16:44.882104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def detect_outliers_iqr(df):\n    outlier_indices = []\n\n    for column in df.select_dtypes(include=['float64', 'int64']).columns:\n        Q1 = df[column].quantile(0.25)\n        Q3 = df[column].quantile(0.75)\n        IQR = Q3 - Q1\n\n        lower_bound = Q1 - 1.5 * IQR\n        upper_bound = Q3 + 1.5 * IQR\n\n        outliers = df[(df[column] < lower_bound) | (df[column] > upper_bound)]\n        outlier_indices.extend(outliers.index)\n\n        print(f'Outliers in {column}:', outliers.shape[0])\n\n    return outlier_indices","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:44.884783Z","iopub.execute_input":"2024-10-11T10:16:44.885175Z","iopub.status.idle":"2024-10-11T10:16:44.901724Z","shell.execute_reply.started":"2024-10-11T10:16:44.885127Z","shell.execute_reply":"2024-10-11T10:16:44.900013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"outlier_indices = detect_outliers_iqr(data)\nprint(\"Total outliers detected:\", len(set(outlier_indices)))","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:44.905513Z","iopub.execute_input":"2024-10-11T10:16:44.906692Z","iopub.status.idle":"2024-10-11T10:16:45.074062Z","shell.execute_reply.started":"2024-10-11T10:16:44.906637Z","shell.execute_reply":"2024-10-11T10:16:45.072735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Analysis","metadata":{}},{"cell_type":"code","source":"data.hist(figsize=(70, 40), bins=30)  \nplt.tight_layout()  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:16:45.075950Z","iopub.execute_input":"2024-10-11T10:16:45.076480Z","iopub.status.idle":"2024-10-11T10:17:06.423237Z","shell.execute_reply.started":"2024-10-11T10:16:45.076407Z","shell.execute_reply":"2024-10-11T10:17:06.421774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## How internet usage affect physical measures ","metadata":{}},{"cell_type":"markdown","source":"### Average BMI , HeartRate , Blood pressure","metadata":{}},{"cell_type":"code","source":"averages = {\n    'BMI': data['Physical-BMI'].mean(),\n    'Heart Rate': data['Physical-HeartRate'].mean(),\n    'Systolic BP': data['Physical-Systolic_BP'].mean(),\n    'Diastolic BP': data['Physical-Diastolic_BP'].mean()\n}\n\n# Convert to DataFrame for easier plotting\naverages_df = pd.DataFrame(list(averages.items()), columns=['Feature', 'Average'])\n\n# Step 2: Create a bar plot\nplt.figure(figsize=(10, 6))\nsns.barplot(x='Feature', y='Average', data=averages_df, palette='coolwarm')\nplt.title('Average Values of BMI, Heart Rate, and Blood Pressure')\nplt.ylabel('Average')\nplt.xlabel('Features')\nplt.ylim(0, averages_df['Average'].max() + 10)  # Adjusting y-axis for better visibility\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:06.424992Z","iopub.execute_input":"2024-10-11T10:17:06.425462Z","iopub.status.idle":"2024-10-11T10:17:06.736649Z","shell.execute_reply.started":"2024-10-11T10:17:06.425389Z","shell.execute_reply":"2024-10-11T10:17:06.734259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Childern Global Assessment Scale vs internet usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.histplot(data['CGAS-CGAS_Score'], bins=30, kde=True)  \nplt.title('Distribution of Childrens Global Assessment Scale Score')\nplt.xlabel('Assessment score')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:06.739147Z","iopub.execute_input":"2024-10-11T10:17:06.739658Z","iopub.status.idle":"2024-10-11T10:17:07.164182Z","shell.execute_reply.started":"2024-10-11T10:17:06.739609Z","shell.execute_reply":"2024-10-11T10:17:07.162894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Assuming 'data' is your DataFrame\nmean_values = data.groupby('sii')['CGAS-CGAS_Score'].mean().reset_index()\n\n# Create a bar plot\nplt.figure(figsize=(10, 8))\nsns.barplot(x='sii', y='CGAS-CGAS_Score', data=mean_values, palette='viridis')\n\n# Customize the plot\nplt.title('Average Fitness Endurance Max Stage vs Total Internet Usage')\nplt.xlabel('Total Internet Usage (hours/day)')\nplt.ylabel('Average Fitness Endurance Max Stage')\nplt.ylim(0, 100)  # Set y-axis limits from 0 to 100\nplt.grid(axis='y')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:07.165891Z","iopub.execute_input":"2024-10-11T10:17:07.166385Z","iopub.status.idle":"2024-10-11T10:17:07.441077Z","shell.execute_reply.started":"2024-10-11T10:17:07.166331Z","shell.execute_reply":"2024-10-11T10:17:07.439975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### BMI vs SII Severity","metadata":{}},{"cell_type":"code","source":"# Plot 1: BMI vs SII Severity\nplt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-BMI', palette='pastel')\nplt.title('BMI vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('BMI')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:07.442325Z","iopub.execute_input":"2024-10-11T10:17:07.442697Z","iopub.status.idle":"2024-10-11T10:17:07.713342Z","shell.execute_reply.started":"2024-10-11T10:17:07.442658Z","shell.execute_reply":"2024-10-11T10:17:07.712172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Blood pressure vs internet Usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-Systolic_BP', palette='pastel')\nplt.title('Blood Pressure vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('Systolic BP')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:07.714863Z","iopub.execute_input":"2024-10-11T10:17:07.715241Z","iopub.status.idle":"2024-10-11T10:17:07.974711Z","shell.execute_reply.started":"2024-10-11T10:17:07.715203Z","shell.execute_reply":"2024-10-11T10:17:07.973488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Heart rate vs internet usage","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nsns.boxplot(data=data, x='sii', y='Physical-HeartRate', palette='pastel')\nplt.title('Heart Rate vs SII Severity')\nplt.xlabel('SII Severity')\nplt.ylabel('Heart Rate')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:07.976472Z","iopub.execute_input":"2024-10-11T10:17:07.976957Z","iopub.status.idle":"2024-10-11T10:17:08.218230Z","shell.execute_reply.started":"2024-10-11T10:17:07.976904Z","shell.execute_reply":"2024-10-11T10:17:08.216657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Demographics influence  ","metadata":{}},{"cell_type":"markdown","source":"### Age Distribution ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(10, 8))\nsns.histplot(data['Basic_Demos-Age'], bins=30, kde=True)  \nplt.title('Distribution of Age Feature')\nplt.xlabel('AGE')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:08.220114Z","iopub.execute_input":"2024-10-11T10:17:08.220676Z","iopub.status.idle":"2024-10-11T10:17:08.615446Z","shell.execute_reply.started":"2024-10-11T10:17:08.220618Z","shell.execute_reply":"2024-10-11T10:17:08.614240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Gender Distribution","metadata":{}},{"cell_type":"code","source":"# Count the number of occurrences for each sex\nGende_counts = data['Basic_Demos-Sex'].value_counts()\n\n# Create the pie chart\nplt.figure(figsize=(7, 7))\nplt.pie(Gende_counts, labels=Gende_counts.index, autopct='%1.1f%%', startangle=90, colors=sns.color_palette('pastel'))\nplt.title('Distribution of Sex in the Dataset')\nplt.axis('equal')  # Equal aspect ratio ensures that pie chart is circular\nplt.show()\n# 0-> male  1_> female","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:08.617097Z","iopub.execute_input":"2024-10-11T10:17:08.617539Z","iopub.status.idle":"2024-10-11T10:17:08.752032Z","shell.execute_reply.started":"2024-10-11T10:17:08.617496Z","shell.execute_reply":"2024-10-11T10:17:08.750536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def categorize_severity(score):\n    if score <= 30:\n        return 'None'\n    elif score <= 49:\n        return 'Mild'\n    elif score <= 79:\n        return 'Moderate'\n    else:\n        return 'Severe'\n\ndata['Severity'] = data['PCIAT-PCIAT_Total'].apply(categorize_severity)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:08.753697Z","iopub.execute_input":"2024-10-11T10:17:08.754174Z","iopub.status.idle":"2024-10-11T10:17:08.763934Z","shell.execute_reply.started":"2024-10-11T10:17:08.754108Z","shell.execute_reply":"2024-10-11T10:17:08.762654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nplt.figure(figsize=(12, 6))\nsns.boxplot(x='Severity', y='Basic_Demos-Age', data=data, palette='pastel')\nplt.title('Age Distribution Across Severity Impairment Index')\nplt.xlabel('Severity Impairment Index')\nplt.ylabel('Age')\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:08.765989Z","iopub.execute_input":"2024-10-11T10:17:08.766484Z","iopub.status.idle":"2024-10-11T10:17:09.034383Z","shell.execute_reply.started":"2024-10-11T10:17:08.766405Z","shell.execute_reply":"2024-10-11T10:17:09.033111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# Calculate the average age for each severity category\naverage_age = data.groupby('Severity')['Basic_Demos-Age'].mean().reset_index()\n\n# Create a line plot\nplt.figure(figsize=(12, 6))\nsns.lineplot(x='Severity', y='Basic_Demos-Age', data=average_age, marker='o')\nplt.title('Average Age Across Severity Impairment Index')\nplt.xlabel('Severity Impairment Index')\nplt.ylabel('Average Age')\nplt.xticks(rotation=45)  # Rotate x labels for better visibility\nplt.grid(True)  # Add grid for better readability\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:09.035805Z","iopub.execute_input":"2024-10-11T10:17:09.036179Z","iopub.status.idle":"2024-10-11T10:17:09.341673Z","shell.execute_reply.started":"2024-10-11T10:17:09.036139Z","shell.execute_reply":"2024-10-11T10:17:09.340537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ninternet_usage = data.groupby(['Basic_Demos-Age', 'PreInt_EduHx-computerinternet_hoursday']).size().unstack(fill_value=0)\n\ninternet_usage = internet_usage.reset_index()\n\ninternet_usage_melted = internet_usage.melt(id_vars='Basic_Demos-Age', \n                                              var_name='Internet Usage', \n                                              value_name='Count')\nplt.figure(figsize=(12, 6))\npalette = sns.color_palette('pastel') \nsns.barplot(data=internet_usage_melted, \n            x='Basic_Demos-Age', \n            y='Count', \n            hue='Internet Usage', \n            palette=palette)\n\nplt.xlabel('Age Group', fontsize=12)\nplt.ylabel('Count', fontsize=12)\nplt.title('Internet Usage Distribution by Age Group', fontsize=14)\nplt.xticks(rotation=45)\n\nhandles = []\nfor i, label in enumerate(['0=Less than 1h/day', '1=Around 1h/day', '2=Around 2hs/day', '3=More than 3hs/day']):\n    handles.append(plt.Line2D([0], [0], color=palette[i], lw=4))  # Create a line for each color\n\nplt.legend(handles, \n           ['0=Less than 1h/day', '1=Around 1h/day', '2=Around 2hs/day', '3=More than 3hs/day'], \n           title='Daily Internet Usage')\n\nplt.tight_layout() \nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:09.343532Z","iopub.execute_input":"2024-10-11T10:17:09.343932Z","iopub.status.idle":"2024-10-11T10:17:10.025782Z","shell.execute_reply.started":"2024-10-11T10:17:09.343885Z","shell.execute_reply":"2024-10-11T10:17:10.024499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sleep Quality vs internet usage","metadata":{}},{"cell_type":"code","source":"mean_sleep_time = data.groupby('PreInt_EduHx-computerinternet_hoursday')['SDS-SDS_Total_T'].mean()\n\nplt.figure(figsize=(10, 8))\nplt.plot(mean_sleep_time.index, mean_sleep_time.values, marker='o', linestyle='-', color='blue')\n\n# Customize the plot\nplt.xlabel('Internet Usage Time (hours/day)')\nplt.ylabel('Average Sleep Time (hours)')\nplt.title('Average Sleep Time vs Internet Usage Time')\n\ncustom_ticks = [ 45, 50, 55, 60, 65, 70,75]  # Define your custom tick values\nplt.yticks(custom_ticks)\n\nplt.grid(True)\n\n# Show the plot\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:10.031889Z","iopub.execute_input":"2024-10-11T10:17:10.032319Z","iopub.status.idle":"2024-10-11T10:17:10.287293Z","shell.execute_reply.started":"2024-10-11T10:17:10.032277Z","shell.execute_reply":"2024-10-11T10:17:10.285881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Behavioural influence ","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n\nplt.figure(figsize=(12, 6))\nsns.boxplot(x='PCIAT-PCIAT_06', y='Basic_Demos-Age', data=data, palette='pastel')\nplt.title('Age Distribution for Q: How often do your childs grades suffer because of the amount of time he or she spends online?')\nplt.xlabel('Response')\nplt.ylabel('Age')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:10.289022Z","iopub.execute_input":"2024-10-11T10:17:10.289539Z","iopub.status.idle":"2024-10-11T10:17:10.595146Z","shell.execute_reply.started":"2024-10-11T10:17:10.289481Z","shell.execute_reply":"2024-10-11T10:17:10.593726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\n# List of questions to analyze\nquestions = [\n    'PCIAT-PCIAT_01', 'PCIAT-PCIAT_02', 'PCIAT-PCIAT_03', \n    'PCIAT-PCIAT_04', 'PCIAT-PCIAT_05', 'PCIAT-PCIAT_06',\n    'PCIAT-PCIAT_07', 'PCIAT-PCIAT_08', 'PCIAT-PCIAT_09',\n    'PCIAT-PCIAT_10', 'PCIAT-PCIAT_11', 'PCIAT-PCIAT_12',\n    'PCIAT-PCIAT_13', 'PCIAT-PCIAT_14', 'PCIAT-PCIAT_15',\n    'PCIAT-PCIAT_16', 'PCIAT-PCIAT_17', 'PCIAT-PCIAT_18',\n    'PCIAT-PCIAT_19', 'PCIAT-PCIAT_20'\n]\n\nplt.figure(figsize=(15, 40))  # Adjust the size for better visibility\n\n# Create a box plot for each question\nfor i, question in enumerate(questions):\n    plt.subplot(len(questions), 1, i + 1)  # Create subplots\n    sns.boxplot(x=question, y='Basic_Demos-Age', data=data, palette='pastel')\n    plt.title(f'Age Distribution for {question}')\n    plt.xlabel('Response')\n    plt.ylabel('Age')\n\nplt.tight_layout()  # Adjust layout to prevent overlap\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:10.596769Z","iopub.execute_input":"2024-10-11T10:17:10.597264Z","iopub.status.idle":"2024-10-11T10:17:16.626717Z","shell.execute_reply.started":"2024-10-11T10:17:10.597198Z","shell.execute_reply":"2024-10-11T10:17:16.625372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Cleaning","metadata":{}},{"cell_type":"markdown","source":"## Handling Missing Values ","metadata":{}},{"cell_type":"markdown","source":"### Drop columns with null>50","metadata":{}},{"cell_type":"code","source":"\ncol = ['CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', \n       'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season',\n       'PCIAT-Season', 'SDS-Season', 'PreInt_EduHx-Season']\ncol.extend( \n        ['Physical-Waist_Circumference','Fitness_Endurance-Max_Stage',\n         'Fitness_Endurance-Time_Mins','Fitness_Endurance-Time_Sec','FGC-FGC_GSND',\n         'FGC-FGC_GSND_Zone','FGC-FGC_GSD','FGC-FGC_GSD_Zone','BIA-BIA_Activity_Level_num',\n         'BIA-BIA_BMC','BIA-BIA_BMI','BIA-BIA_BMR','BIA-BIA_DEE','BIA-BIA_ECW','BIA-BIA_FFM',\n         'BIA-BIA_FFMI','BIA-BIA_Fat','BIA-BIA_Frame_num','BIA-BIA_ICW','BIA-BIA_LDM','BIA-BIA_LST',\n         'BIA-BIA_SMM','BIA-BIA_TBW','PAQ_A-PAQ_A_Total','PAQ_C-PAQ_C_Total'])\nlen(col)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:16.628676Z","iopub.execute_input":"2024-10-11T10:17:16.629163Z","iopub.status.idle":"2024-10-11T10:17:16.642697Z","shell.execute_reply.started":"2024-10-11T10:17:16.629112Z","shell.execute_reply":"2024-10-11T10:17:16.640845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.drop(col , axis =1 ,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:16.644381Z","iopub.execute_input":"2024-10-11T10:17:16.644797Z","iopub.status.idle":"2024-10-11T10:17:16.667594Z","shell.execute_reply.started":"2024-10-11T10:17:16.644754Z","shell.execute_reply":"2024-10-11T10:17:16.666216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:16.669161Z","iopub.execute_input":"2024-10-11T10:17:16.670227Z","iopub.status.idle":"2024-10-11T10:17:16.684720Z","shell.execute_reply.started":"2024-10-11T10:17:16.670147Z","shell.execute_reply":"2024-10-11T10:17:16.683531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_cleaned= data.copy()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:16.686146Z","iopub.execute_input":"2024-10-11T10:17:16.686573Z","iopub.status.idle":"2024-10-11T10:17:16.701262Z","shell.execute_reply.started":"2024-10-11T10:17:16.686524Z","shell.execute_reply":"2024-10-11T10:17:16.699769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_cleaned.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:16.703027Z","iopub.execute_input":"2024-10-11T10:17:16.703531Z","iopub.status.idle":"2024-10-11T10:17:16.715933Z","shell.execute_reply.started":"2024-10-11T10:17:16.703484Z","shell.execute_reply":"2024-10-11T10:17:16.714619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Filling Missing Values in first 46 columns with Knn imputer","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\ndata_cleaned['Basic_Demos-Enroll_Season'] = label_encoder.fit_transform(data_cleaned['Basic_Demos-Enroll_Season'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:16.717639Z","iopub.execute_input":"2024-10-11T10:17:16.718341Z","iopub.status.idle":"2024-10-11T10:17:16.732211Z","shell.execute_reply.started":"2024-10-11T10:17:16.718285Z","shell.execute_reply":"2024-10-11T10:17:16.730748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import KNNImputer\n\nimputer = KNNImputer(n_neighbors=5)\ndata_cleaned.iloc[:, :45] = imputer.fit_transform(data_cleaned.iloc[:, :45])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:16.734080Z","iopub.execute_input":"2024-10-11T10:17:16.734630Z","iopub.status.idle":"2024-10-11T10:17:21.527513Z","shell.execute_reply.started":"2024-10-11T10:17:16.734581Z","shell.execute_reply":"2024-10-11T10:17:21.526263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nlabel_encoder = LabelEncoder()\ndata_cleaned['Severity'] = label_encoder.fit_transform(data_cleaned['Severity'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:21.529138Z","iopub.execute_input":"2024-10-11T10:17:21.529542Z","iopub.status.idle":"2024-10-11T10:17:21.536805Z","shell.execute_reply.started":"2024-10-11T10:17:21.529501Z","shell.execute_reply":"2024-10-11T10:17:21.535383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Filling missing values in target with Kmeans","metadata":{}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nimport numpy as np\nimport pandas as pd\n\ndef impute_with_kmeans(df, categorical_columns, n_clusters=5):\n    # Fill missing values temporarily with the mode (or any placeholder)\n    df_temp = df.copy()\n    for col in categorical_columns:\n        df_temp[col].fillna(df_temp[col].mode()[0], inplace=True)\n\n    # Perform KMeans clustering after filling missing values\n    kmeans = KMeans(n_clusters=n_clusters, random_state=0)\n    cluster_labels = kmeans.fit_predict(df_temp)\n\n    # Impute missing values within each cluster\n    for col in categorical_columns:\n        for cluster in np.unique(cluster_labels):\n            mask = (cluster_labels == cluster) & df[col].isna()\n            most_frequent = df.loc[cluster_labels == cluster, col].mode()[0]\n            df.loc[mask, col] = most_frequent\n\n    return df\n\n# Apply KMeans-based imputation\ndata_cleaned = impute_with_kmeans(data_cleaned, ['sii'])","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:21.538071Z","iopub.execute_input":"2024-10-11T10:17:21.538484Z","iopub.status.idle":"2024-10-11T10:17:23.071299Z","shell.execute_reply.started":"2024-10-11T10:17:21.538420Z","shell.execute_reply":"2024-10-11T10:17:23.069822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_cleaned['sii'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:23.072905Z","iopub.execute_input":"2024-10-11T10:17:23.073303Z","iopub.status.idle":"2024-10-11T10:17:23.082957Z","shell.execute_reply.started":"2024-10-11T10:17:23.073262Z","shell.execute_reply":"2024-10-11T10:17:23.081622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_cleaned.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:23.084402Z","iopub.execute_input":"2024-10-11T10:17:23.084808Z","iopub.status.idle":"2024-10-11T10:17:23.119866Z","shell.execute_reply.started":"2024-10-11T10:17:23.084768Z","shell.execute_reply":"2024-10-11T10:17:23.118274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_cleaned.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:23.121709Z","iopub.execute_input":"2024-10-11T10:17:23.122072Z","iopub.status.idle":"2024-10-11T10:17:23.132583Z","shell.execute_reply.started":"2024-10-11T10:17:23.122037Z","shell.execute_reply":"2024-10-11T10:17:23.131154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Handling outliers ","metadata":{}},{"cell_type":"code","source":"outlier_indices = detect_outliers_iqr(data_cleaned)\nprint(\"Total outliers detected:\", len(set(outlier_indices)))","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:23.134032Z","iopub.execute_input":"2024-10-11T10:17:23.134450Z","iopub.status.idle":"2024-10-11T10:17:23.256932Z","shell.execute_reply.started":"2024-10-11T10:17:23.134384Z","shell.execute_reply":"2024-10-11T10:17:23.255468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for column in data_cleaned.columns:\n    plt.figure(figsize=(8, 3))  \n    sns.boxplot(data=data_cleaned[[column]], orient='h', palette=\"Set2\")\n    \n    plt.title(f'Box Plot for {column} with Outliers', fontsize=16)\n    plt.ylabel(column, fontsize=14)\n    plt.xlabel('Values', fontsize=14)\n    \n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:23.258419Z","iopub.execute_input":"2024-10-11T10:17:23.258853Z","iopub.status.idle":"2024-10-11T10:17:33.936654Z","shell.execute_reply.started":"2024-10-11T10:17:23.258810Z","shell.execute_reply":"2024-10-11T10:17:33.935343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate IQR for the column 'CGAS-CGAS_Score'\nQ1 = data_cleaned['CGAS-CGAS_Score'].quantile(0.25)\nQ3 = data_cleaned['CGAS-CGAS_Score'].quantile(0.75)\nIQR = Q3 - Q1\n\n# Define outlier boundaries\nlower_bound = Q1 - 1.5 * IQR\nupper_bound = Q3 + 1.5 * IQR\n\n# Filter the entire DataFrame by keeping only rows where 'CGAS-CGAS_Score' is within bounds\ndata_copy = data_cleaned[(data_cleaned['CGAS-CGAS_Score'] >= lower_bound) & (data_cleaned['CGAS-CGAS_Score'] <= upper_bound)]\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:33.938138Z","iopub.execute_input":"2024-10-11T10:17:33.938520Z","iopub.status.idle":"2024-10-11T10:17:33.951370Z","shell.execute_reply.started":"2024-10-11T10:17:33.938477Z","shell.execute_reply":"2024-10-11T10:17:33.950050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"markdown","source":"## BMI To Heart rate Ratio","metadata":{}},{"cell_type":"code","source":"data_copy['HeartRate_BMI'] = data_copy['Physical-HeartRate'] * data_copy['Physical-BMI']","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:33.952916Z","iopub.execute_input":"2024-10-11T10:17:33.953320Z","iopub.status.idle":"2024-10-11T10:17:33.964766Z","shell.execute_reply.started":"2024-10-11T10:17:33.953279Z","shell.execute_reply":"2024-10-11T10:17:33.963315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Max Heart rate due to age","metadata":{}},{"cell_type":"code","source":"data_copy['HRmax'] = 220 - data_copy['Basic_Demos-Age']  # Estimate HRmax based on age","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:33.966537Z","iopub.execute_input":"2024-10-11T10:17:33.967084Z","iopub.status.idle":"2024-10-11T10:17:33.984611Z","shell.execute_reply.started":"2024-10-11T10:17:33.967023Z","shell.execute_reply":"2024-10-11T10:17:33.983326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Body Strength & Flexibility","metadata":{}},{"cell_type":"code","source":"data_copy['BodyStrength&Flexibility'] = data_copy['FGC-FGC_CU'] + data_copy['FGC-FGC_PU'] + data_copy['FGC-FGC_SRL'] + data_copy['FGC-FGC_SRR'] + data_copy['FGC-FGC_TL']\ndata_copy['BodyStrength&Flexibility_class'] = data_copy['FGC-FGC_CU_Zone'] + data_copy['FGC-FGC_PU_Zone'] + data_copy['FGC-FGC_SRL_Zone'] + data_copy['FGC-FGC_SRR_Zone'] + data_copy['FGC-FGC_TL_Zone']\n\ndata_copy = data_copy.drop(['FGC-FGC_CU','FGC-FGC_PU','FGC-FGC_SRL','FGC-FGC_SRR','FGC-FGC_TL'], axis=1)\n\ndata_copy = data_copy.drop(['FGC-FGC_CU_Zone','FGC-FGC_PU_Zone','FGC-FGC_SRL_Zone','FGC-FGC_SRR_Zone','FGC-FGC_TL_Zone'], axis=1)                          \n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:33.986161Z","iopub.execute_input":"2024-10-11T10:17:33.986710Z","iopub.status.idle":"2024-10-11T10:17:34.004521Z","shell.execute_reply.started":"2024-10-11T10:17:33.986653Z","shell.execute_reply":"2024-10-11T10:17:34.003271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_copy.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:34.006302Z","iopub.execute_input":"2024-10-11T10:17:34.006730Z","iopub.status.idle":"2024-10-11T10:17:34.019554Z","shell.execute_reply.started":"2024-10-11T10:17:34.006686Z","shell.execute_reply":"2024-10-11T10:17:34.018040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transfomation and scaling","metadata":{}},{"cell_type":"markdown","source":"## Data Splitting ","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = data_copy.drop('sii', axis=1)  \ny = data_copy['sii']  \nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)\nprint(X_train.shape, X_test.shape, y_train.shape, y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:34.021187Z","iopub.execute_input":"2024-10-11T10:17:34.021714Z","iopub.status.idle":"2024-10-11T10:17:34.039598Z","shell.execute_reply.started":"2024-10-11T10:17:34.021656Z","shell.execute_reply":"2024-10-11T10:17:34.038309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Handling imbalanced data ","metadata":{}},{"cell_type":"code","source":"from collections import Counter\nfrom imblearn.over_sampling import SMOTE\n\n# Check the class distribution before applying SMOTE\nclass_distribution_before = Counter(y_train)\nprint(\"Class distribution before SMOTE:\", class_distribution_before)\n\n# Initialize SMOTE with specified parameters\nsmote = SMOTE(sampling_strategy='auto', random_state=42)\n\n# Apply SMOTE to the training data\nX_train_resampled, y_train_resampled = smote.fit_resample(X_train, y_train)\n\n# Check the class distribution after applying SMOTE\nclass_distribution_after = Counter(y_train_resampled)\nprint(\"Class distribution after SMOTE:\", class_distribution_after)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:34.041111Z","iopub.execute_input":"2024-10-11T10:17:34.041539Z","iopub.status.idle":"2024-10-11T10:17:34.081684Z","shell.execute_reply.started":"2024-10-11T10:17:34.041494Z","shell.execute_reply":"2024-10-11T10:17:34.080350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scaling","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import RobustScaler\nscaler = RobustScaler()  # or StandardScaler()\n\n# Fit the scaler on X_train and transform both X_train and X_test\nX_train = pd.DataFrame(scaler.fit_transform(X_train_resampled), columns=X_train_resampled.columns)\nX_test = pd.DataFrame(scaler.transform(X_test), columns=X_test.columns)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:34.083233Z","iopub.execute_input":"2024-10-11T10:17:34.083646Z","iopub.status.idle":"2024-10-11T10:17:34.119241Z","shell.execute_reply.started":"2024-10-11T10:17:34.083604Z","shell.execute_reply":"2024-10-11T10:17:34.118044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:34.121018Z","iopub.execute_input":"2024-10-11T10:17:34.121538Z","iopub.status.idle":"2024-10-11T10:17:34.129384Z","shell.execute_reply.started":"2024-10-11T10:17:34.121481Z","shell.execute_reply":"2024-10-11T10:17:34.128336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:34.131211Z","iopub.execute_input":"2024-10-11T10:17:34.131603Z","iopub.status.idle":"2024-10-11T10:17:34.143845Z","shell.execute_reply.started":"2024-10-11T10:17:34.131564Z","shell.execute_reply":"2024-10-11T10:17:34.142681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling ","metadata":{}},{"cell_type":"code","source":"result1, result2, result3 = [], [], [] ","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:34.145089Z","iopub.execute_input":"2024-10-11T10:17:34.145510Z","iopub.status.idle":"2024-10-11T10:17:34.155963Z","shell.execute_reply.started":"2024-10-11T10:17:34.145465Z","shell.execute_reply":"2024-10-11T10:17:34.154540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def modeling(model):\n    # Fit the model using the scaled resampled training data\n    model.fit(X_train, y_train_resampled)  # Use y_train_resampled here\n    \n    # Predictions\n    train_pred = model.predict(X_train)    # Predictions on the scaled resampled train set\n    test_pred = model.predict(X_test)      # Predictions on the scaled original test set\n    \n    # Calculate metrics with specified average\n    train_accuracy = accuracy_score(y_train_resampled, train_pred) * 100\n    train_recall = recall_score(y_train_resampled, train_pred, average='weighted') * 100\n    train_f1_score = f1_score(y_train_resampled, train_pred, average='weighted') * 100\n    \n    test_accuracy = accuracy_score(y_test, test_pred) * 100\n    test_recall = recall_score(y_test, test_pred, average='weighted') * 100\n    test_f1_score = f1_score(y_test, test_pred, average='weighted') * 100\n    \n    # Append results\n    result1.append(test_accuracy)\n    result2.append(test_recall)\n    result3.append(test_f1_score)\n    \n    print(\"Classification Report for Test Data:\")\n    print(classification_report(y_test, test_pred))\n    \n    print(\"\\nClassification Report for Scaled Resampled Train Data:\")\n    print(classification_report(y_train_resampled, train_pred))  # Use y_train_resampled here\n    \n    # Accuracy, Recall, and F1 Scores\n    print(f'Training Accuracy: {train_accuracy}, Train Recall: {train_recall}, Train F1: {train_f1_score}')\n    print(f'Test Accuracy: {test_accuracy}, Test Recall: {test_recall}, Test F1: {test_f1_score}')\n    \n    # Confusion matrix\n    cm = confusion_matrix(y_test, test_pred)\n    sns.heatmap(cm, annot=True, fmt='0.2f', cmap='YlGnBu', linewidths=1)\n    plt.xlabel('Predicted')\n    plt.ylabel('Actual')\n    plt.title('Confusion Matrix')\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:34.170860Z","iopub.execute_input":"2024-10-11T10:17:34.171346Z","iopub.status.idle":"2024-10-11T10:17:34.187034Z","shell.execute_reply.started":"2024-10-11T10:17:34.171303Z","shell.execute_reply":"2024-10-11T10:17:34.185633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression  \nlogistic_regression = LogisticRegression()\nmodeling(logistic_regression)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:34.188575Z","iopub.execute_input":"2024-10-11T10:17:34.189010Z","iopub.status.idle":"2024-10-11T10:17:34.892881Z","shell.execute_reply.started":"2024-10-11T10:17:34.188967Z","shell.execute_reply":"2024-10-11T10:17:34.891467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SVM = SVC()\nmodeling(SVM)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:17:34.894303Z","iopub.execute_input":"2024-10-11T10:17:34.894694Z","iopub.status.idle":"2024-10-11T10:17:36.343812Z","shell.execute_reply.started":"2024-10-11T10:17:34.894651Z","shell.execute_reply":"2024-10-11T10:17:36.342660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"decision_tree_classifier = DecisionTreeClassifier()\nmodeling(decision_tree_classifier)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:26:30.997107Z","iopub.execute_input":"2024-10-11T10:26:30.997679Z","iopub.status.idle":"2024-10-11T10:26:31.455418Z","shell.execute_reply.started":"2024-10-11T10:26:30.997627Z","shell.execute_reply":"2024-10-11T10:26:31.454068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"decision_tree_classifier = DecisionTreeClassifier( max_depth= 6 , min_samples_split= 6)\nmodeling(decision_tree_classifier)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:26:04.806539Z","iopub.execute_input":"2024-10-11T10:26:04.807037Z","iopub.status.idle":"2024-10-11T10:26:05.273653Z","shell.execute_reply.started":"2024-10-11T10:26:04.806996Z","shell.execute_reply":"2024-10-11T10:26:05.272348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sgd_classifier = SGDClassifier(max_iter = 500) \nmodeling(sgd_classifier)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:27:07.419864Z","iopub.execute_input":"2024-10-11T10:27:07.420386Z","iopub.status.idle":"2024-10-11T10:27:08.004500Z","shell.execute_reply.started":"2024-10-11T10:27:07.420340Z","shell.execute_reply":"2024-10-11T10:27:08.002826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nknn = KNeighborsClassifier()\nmodeling(knn)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:27:53.963552Z","iopub.execute_input":"2024-10-11T10:27:53.964634Z","iopub.status.idle":"2024-10-11T10:27:55.016136Z","shell.execute_reply.started":"2024-10-11T10:27:53.964584Z","shell.execute_reply":"2024-10-11T10:27:55.014806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n\n# Create an instance of GaussianNB\nnaive_bayes = GaussianNB()\n\n# Call your modeling function\nmodeling(naive_bayes)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:28:55.980190Z","iopub.execute_input":"2024-10-11T10:28:55.980670Z","iopub.status.idle":"2024-10-11T10:28:56.381895Z","shell.execute_reply.started":"2024-10-11T10:28:55.980612Z","shell.execute_reply":"2024-10-11T10:28:56.380667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nrandom_forest = RandomForestClassifier(random_state=1)\nmodeling(random_forest)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:29:38.846579Z","iopub.execute_input":"2024-10-11T10:29:38.847026Z","iopub.status.idle":"2024-10-11T10:29:40.983518Z","shell.execute_reply.started":"2024-10-11T10:29:38.846982Z","shell.execute_reply":"2024-10-11T10:29:40.982088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"adaBosster=AdaBoostClassifier( n_estimators=50)\nmodeling(adaBosster)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:29:56.326727Z","iopub.execute_input":"2024-10-11T10:29:56.328291Z","iopub.status.idle":"2024-10-11T10:29:58.244955Z","shell.execute_reply.started":"2024-10-11T10:29:56.328221Z","shell.execute_reply":"2024-10-11T10:29:58.243678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingClassifier\n\ngradient=GradientBoostingClassifier()\nmodeling(gradient)","metadata":{"execution":{"iopub.status.busy":"2024-10-11T10:30:59.149957Z","iopub.execute_input":"2024-10-11T10:30:59.150389Z","iopub.status.idle":"2024-10-11T10:31:22.475512Z","shell.execute_reply.started":"2024-10-11T10:30:59.150348Z","shell.execute_reply":"2024-10-11T10:31:22.474295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}