{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30823,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Import Important Libraries","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np \nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split, GridSearchCV\nfrom sklearn.impute import SimpleImputer, KNNImputer\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder, OneHotEncoder\n\nimport warnings\nwarnings.filterwarnings('ignore', category=FutureWarning)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:02:52.601450Z","iopub.execute_input":"2024-12-19T21:02:52.601726Z","iopub.status.idle":"2024-12-19T21:02:54.492364Z","shell.execute_reply.started":"2024-12-19T21:02:52.601703Z","shell.execute_reply":"2024-12-19T21:02:54.491711Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load essential dataset","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\ndf_test = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/test.csv')\ndf_dict = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/data_dictionary.csv')\ndf_sub_samp = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/sample_submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:02:59.064056Z","iopub.execute_input":"2024-12-19T21:02:59.064470Z","iopub.status.idle":"2024-12-19T21:02:59.134801Z","shell.execute_reply.started":"2024-12-19T21:02:59.064445Z","shell.execute_reply":"2024-12-19T21:02:59.134215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_sub_samp.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:02:59.135849Z","iopub.execute_input":"2024-12-19T21:02:59.136121Z","iopub.status.idle":"2024-12-19T21:02:59.149525Z","shell.execute_reply.started":"2024-12-19T21:02:59.136099Z","shell.execute_reply":"2024-12-19T21:02:59.148689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# drop id column for train dataset\ndf_train = df_train.drop(['id'], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:02:59.978210Z","iopub.execute_input":"2024-12-19T21:02:59.978559Z","iopub.status.idle":"2024-12-19T21:02:59.989122Z","shell.execute_reply.started":"2024-12-19T21:02:59.978531Z","shell.execute_reply":"2024-12-19T21:02:59.988044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# drop id column for test dataset\ndf_test = df_test.drop(['id'], axis = 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:00.346953Z","iopub.execute_input":"2024-12-19T21:03:00.347254Z","iopub.status.idle":"2024-12-19T21:03:00.351413Z","shell.execute_reply.started":"2024-12-19T21:03:00.347231Z","shell.execute_reply":"2024-12-19T21:03:00.350495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:01.282935Z","iopub.execute_input":"2024-12-19T21:03:01.283276Z","iopub.status.idle":"2024-12-19T21:03:01.306284Z","shell.execute_reply.started":"2024-12-19T21:03:01.283235Z","shell.execute_reply":"2024-12-19T21:03:01.305306Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"markdown","source":"# Checking missing answers in label column","metadata":{}},{"cell_type":"markdown","source":"I strongly recommend checking out this notebook: https://www.kaggle.com/code/antoninadolgorukova/cmi-piu-features-eda#Features-EDA-by-Groups. It offers a thorough exploratory data analysis (EDA) grouped by feature categories and is a great resource for insights into the dataset.","metadata":{}},{"cell_type":"markdown","source":"In the PCIAT columns, there are questions related to the label column. If any of these questions have missing values, it could impact the label column's accuracy. To address this, we will remove rows containing missing values in these columns to ensure data integrity.","metadata":{}},{"cell_type":"code","source":"PCIAT_cols = list(set(df_train.columns) - set(df_test.columns))\nPCIAT_cols","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:03.090356Z","iopub.execute_input":"2024-12-19T21:03:03.090635Z","iopub.status.idle":"2024-12-19T21:03:03.096060Z","shell.execute_reply.started":"2024-12-19T21:03:03.090613Z","shell.execute_reply":"2024-12-19T21:03:03.095277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_cols = set(df_train.columns)\ntest_cols = set(df_test.columns)\ncolumns_not_in_test = sorted(list(train_cols - test_cols))\ndf_dict[df_dict['Field'].isin(columns_not_in_test)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:05.024537Z","iopub.execute_input":"2024-12-19T21:03:05.024811Z","iopub.status.idle":"2024-12-19T21:03:05.039591Z","shell.execute_reply.started":"2024-12-19T21:03:05.024790Z","shell.execute_reply":"2024-12-19T21:03:05.038645Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pciat_min_max = df_train.groupby('sii')['PCIAT-PCIAT_Total'].agg(['min', 'max'])\npciat_min_max = pciat_min_max.rename(\n    columns={'min': 'Minimum PCIAT total Score', 'max': 'Maximum total PCIAT Score'}\n)\npciat_min_max","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:05.460536Z","iopub.execute_input":"2024-12-19T21:03:05.460809Z","iopub.status.idle":"2024-12-19T21:03:05.477602Z","shell.execute_reply.started":"2024-12-19T21:03:05.460788Z","shell.execute_reply":"2024-12-19T21:03:05.476853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_dict[df_dict['Field'] == 'PCIAT-PCIAT_Total']['Value Labels'].iloc[0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:06.783247Z","iopub.execute_input":"2024-12-19T21:03:06.783563Z","iopub.status.idle":"2024-12-19T21:03:06.789749Z","shell.execute_reply.started":"2024-12-19T21:03:06.783541Z","shell.execute_reply":"2024-12-19T21:03:06.788892Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_with_sii = df_train[df_train['sii'].notna()][columns_not_in_test]\ntrain_with_sii[train_with_sii.isna().any(axis=1)].head().style.applymap(\n    lambda x: 'background-color: #FFC0CB' if pd.isna(x) else ''\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:07.070481Z","iopub.execute_input":"2024-12-19T21:03:07.070736Z","iopub.status.idle":"2024-12-19T21:03:07.141699Z","shell.execute_reply.started":"2024-12-19T21:03:07.070715Z","shell.execute_reply":"2024-12-19T21:03:07.141030Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"PCIAT_cols = [f'PCIAT-PCIAT_{i+1:02d}' for i in range(20)]\nrecalc_total_score = train_with_sii[PCIAT_cols].sum(\n    axis=1, skipna=True\n)\n(recalc_total_score == train_with_sii['PCIAT-PCIAT_Total']).all()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:09.067118Z","iopub.execute_input":"2024-12-19T21:03:09.067536Z","iopub.status.idle":"2024-12-19T21:03:09.076547Z","shell.execute_reply.started":"2024-12-19T21:03:09.067512Z","shell.execute_reply":"2024-12-19T21:03:09.075491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def recalculate_sii(row):\n    if pd.isna(row['PCIAT-PCIAT_Total']):\n        return np.nan\n    max_possible = row['PCIAT-PCIAT_Total'] + row[PCIAT_cols].isna().sum() * 5\n    if row['PCIAT-PCIAT_Total'] <= 30 and max_possible <= 30:\n        return 0\n    elif 31 <= row['PCIAT-PCIAT_Total'] <= 49 and max_possible <= 49:\n        return 1\n    elif 50 <= row['PCIAT-PCIAT_Total'] <= 79 and max_possible <= 79:\n        return 2\n    elif row['PCIAT-PCIAT_Total'] >= 80 and max_possible >= 80:\n        return 3\n    return np.nan\n\ndf_train['recalc_sii'] = df_train.apply(recalculate_sii, axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:09.636327Z","iopub.execute_input":"2024-12-19T21:03:09.636595Z","iopub.status.idle":"2024-12-19T21:03:10.891384Z","shell.execute_reply.started":"2024-12-19T21:03:09.636575Z","shell.execute_reply":"2024-12-19T21:03:10.890683Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"mismatch_rows = df_train[\n    (df_train['recalc_sii'] != df_train['sii']) & df_train['sii'].notna()\n]\n\nmismatch_rows[PCIAT_cols + [\n    'PCIAT-PCIAT_Total', 'sii', 'recalc_sii'\n]].style.applymap(\n    lambda x: 'background-color: #FFC0CB' if pd.isna(x) else ''\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:10.892446Z","iopub.execute_input":"2024-12-19T21:03:10.892738Z","iopub.status.idle":"2024-12-19T21:03:10.911958Z","shell.execute_reply.started":"2024-12-19T21:03:10.892712Z","shell.execute_reply":"2024-12-19T21:03:10.911242Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"for 17 rows the target variable was calculated incorrectly (ignoring missing responses).\r\n","metadata":{}},{"cell_type":"code","source":"# identify the new label column after dropping the missing answers\ndf_train['sii'] = df_train['recalc_sii']\ndf_train['complete_resp_total'] = df_train['PCIAT-PCIAT_Total'].where(\n    df_train[PCIAT_cols].notna().all(axis=1), np.nan\n)\n\nsii_map = {0: '0 (None)', 1: '1 (Mild)', 2: '2 (Moderate)', 3: '3 (Severe)'}\ndf_train['sii'] = df_train['sii'].map(sii_map).fillna('Missing')\n\nsii_order = ['Missing', '0 (None)', '1 (Mild)', '2 (Moderate)', '3 (Severe)']\ndf_train['sii'] = pd.Categorical(df_train['sii'], categories=sii_order, ordered=True)\n\ndf_train.drop(columns='recalc_sii', inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:21.204237Z","iopub.execute_input":"2024-12-19T21:03:21.204540Z","iopub.status.idle":"2024-12-19T21:03:21.216552Z","shell.execute_reply.started":"2024-12-19T21:03:21.204519Z","shell.execute_reply":"2024-12-19T21:03:21.215784Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"counts_sii = df_train['sii'].value_counts()\ncounts_sii","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:22.472810Z","iopub.execute_input":"2024-12-19T21:03:22.473160Z","iopub.status.idle":"2024-12-19T21:03:22.482020Z","shell.execute_reply.started":"2024-12-19T21:03:22.473130Z","shell.execute_reply":"2024-12-19T21:03:22.481258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distrubition of sii\nax = counts_sii.plot(kind = 'bar')\nplt.title('Distrubition of sii')\nplt.xlabel('sii')\nplt.ylabel('Count')\n\n# Add the exact count on top of each bar \nfor i in ax.containers :\n    ax.bar_label(i, label_type = 'edge')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:22.981257Z","iopub.execute_input":"2024-12-19T21:03:22.981563Z","iopub.status.idle":"2024-12-19T21:03:23.261421Z","shell.execute_reply.started":"2024-12-19T21:03:22.981537Z","shell.execute_reply":"2024-12-19T21:03:23.260514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"enroll_season = df_train['Basic_Demos-Enroll_Season'].value_counts()\nenroll_season","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T23:44:24.111923Z","iopub.execute_input":"2024-12-17T23:44:24.112664Z","iopub.status.idle":"2024-12-17T23:44:24.119809Z","shell.execute_reply.started":"2024-12-17T23:44:24.112628Z","shell.execute_reply":"2024-12-17T23:44:24.118910Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Season Enrollment\nplt.figure(figsize=(8, 6))\nplt.pie(enroll_season, labels=None, autopct = '%1.1f%%', colors = ['#ADD8E6', '#FFA500','#F6F6F6', '#808000'], startangle = 90)\nplt.legend(enroll_season.index, title = \"Enrollment Season\", loc = \"best\")\nplt.title('Demographics Season Enrollment')\nplt.axis('equal')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T23:44:27.317779Z","iopub.execute_input":"2024-12-17T23:44:27.318624Z","iopub.status.idle":"2024-12-17T23:44:27.528118Z","shell.execute_reply.started":"2024-12-17T23:44:27.318586Z","shell.execute_reply":"2024-12-17T23:44:27.526920Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(figsize=(12, 5))\n\n# Age Distribution by Sex\nsns.histplot(\n    data = df_train, x = 'Basic_Demos-Age',\n    hue = 'Basic_Demos-Sex', multiple = 'dodge',\n    palette = \"Set2\", bins = 20, ax = axes\n)\naxes.set_title('Age Distribution by Sex')\naxes.set_xlabel('Age')\naxes.set_ylabel('Count')\n\n# plt.legend(Basic_Demos-Sex.index, title = \"Enrollment Sex\", loc = \"best\")\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T23:44:29.616070Z","iopub.execute_input":"2024-12-17T23:44:29.616877Z","iopub.status.idle":"2024-12-17T23:44:30.197431Z","shell.execute_reply.started":"2024-12-17T23:44:29.616845Z","shell.execute_reply":"2024-12-17T23:44:30.196603Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Physical_enroll_season = df_train['Physical-Season'].value_counts()\nPhysical_enroll_season","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T23:44:34.246162Z","iopub.execute_input":"2024-12-17T23:44:34.246996Z","iopub.status.idle":"2024-12-17T23:44:34.254392Z","shell.execute_reply.started":"2024-12-17T23:44:34.246962Z","shell.execute_reply":"2024-12-17T23:44:34.253433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Physical Season Enrollment\nplt.figure(figsize=(8, 6))\nplt.pie(Physical_enroll_season, labels=None, autopct = '%1.1f%%', colors = ['#ADD8E6', '#FFA500','#F6F6F6', '#808000'], startangle = 90)\nplt.legend(enroll_season.index, title = \"Enrollment Season\", loc = \"best\")\nplt.title('Physical Season Enrollment')\nplt.axis('equal')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T23:44:36.160322Z","iopub.execute_input":"2024-12-17T23:44:36.160684Z","iopub.status.idle":"2024-12-17T23:44:36.369850Z","shell.execute_reply.started":"2024-12-17T23:44:36.160653Z","shell.execute_reply":"2024-12-17T23:44:36.368787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set a style for the plot\nsns.set(style=\"whitegrid\")\n\n# Create a scatter plot\nplt.figure(figsize=(8, 6))\nsns.scatterplot(x='Physical-Height', y='Physical-Weight', data=df_train, color='blue', s=100)\n\n# Add labels and a title\nplt.title(\"Relationship Between Height and Weight\")\nplt.xlabel(\"Height (inches)\")\nplt.ylabel(\"Weight (lbs)\")\n\n# Show the plot\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T23:44:58.050270Z","iopub.execute_input":"2024-12-17T23:44:58.050613Z","iopub.status.idle":"2024-12-17T23:44:58.366853Z","shell.execute_reply.started":"2024-12-17T23:44:58.050583Z","shell.execute_reply":"2024-12-17T23:44:58.366039Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\r\nBlood Pressure vs Heart Rate","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 5))\n\n# Diastolic BP vs Heart Rate\nplt.subplot(1, 2, 1)\nsns.scatterplot(x = 'Physical-Diastolic_BP', y = 'Physical-HeartRate', data = df_train)\nplt.title('Diastolic BP vs Heart Rate')\nplt.xlabel('Diastolic Blood Pressure (mmHg)')\nplt.ylabel('Heart rate (beats/min)')\n\n# Systolic BP vs Heart Rate\nplt.subplot(1, 2, 2)\nsns.scatterplot(x = 'Physical-Systolic_BP', y = 'Physical-HeartRate', data = df_train)\nplt.title('Systolic BP vs Heart Rate')\nplt.xlabel('Systolic Blood Pressure (mmHg)')\nplt.ylabel('Heart rate (beats/min)')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T23:45:00.680573Z","iopub.execute_input":"2024-12-17T23:45:00.681422Z","iopub.status.idle":"2024-12-17T23:45:01.384790Z","shell.execute_reply.started":"2024-12-17T23:45:00.681387Z","shell.execute_reply":"2024-12-17T23:45:01.383969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T22:43:12.659428Z","iopub.execute_input":"2024-12-17T22:43:12.659786Z","iopub.status.idle":"2024-12-17T22:43:12.666058Z","shell.execute_reply.started":"2024-12-17T22:43:12.659754Z","shell.execute_reply":"2024-12-17T22:43:12.665143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distribution of Daily Computer/Internet Usage\n# sns.histplot(df_train['PreInt_EduHx-computerinternet_hoursday'], bins=10, kde=True)\n# plt.title('Distribution of Daily Computer/Internet Usage')\n# plt.xlabel('Hours per Day')\n# plt.ylabel('Frequency')\n# plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create the box plot\n# plt.figure(figsize=(10, 6))\n# sns.boxplot(\n#    x=df_train['sii'],\n#    y=df_train['PreInt_EduHx-computerinternet_hoursday'],\n#    palette='coolwarm'\n#)\n\n## Add labels and title\n# plt.title(\"Internet Usage by SII\")\n# plt.xlabel(\"SII Category\")\n# plt.ylabel(\"Computer Internet Hours/Day\")\n\n## Show the plot\n# plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Engineering","metadata":{}},{"cell_type":"code","source":"y = df_train[\"sii\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:38.719750Z","iopub.execute_input":"2024-12-19T21:03:38.720091Z","iopub.status.idle":"2024-12-19T21:03:38.723820Z","shell.execute_reply.started":"2024-12-19T21:03:38.720060Z","shell.execute_reply":"2024-12-19T21:03:38.723099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y.isna().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:40.038568Z","iopub.execute_input":"2024-12-19T21:03:40.038849Z","iopub.status.idle":"2024-12-19T21:03:40.044186Z","shell.execute_reply.started":"2024-12-19T21:03:40.038828Z","shell.execute_reply":"2024-12-19T21:03:40.043469Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Align columns in train and test set","metadata":{}},{"cell_type":"code","source":"# Align columns in train and test sets\ncommon_cols = list(set(df_train.columns) & set(df_test.columns))\ndf_train = df_train[common_cols]\ndf_test = df_test[common_cols]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:41.036150Z","iopub.execute_input":"2024-12-19T21:03:41.036487Z","iopub.status.idle":"2024-12-19T21:03:41.043444Z","shell.execute_reply.started":"2024-12-19T21:03:41.036459Z","shell.execute_reply":"2024-12-19T21:03:41.042737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Transfer units to cm and kg for df_train\nlbs_to_kg = 0.453592\ninches_to_cm = 2.54\n\ndf_train['Physical-Height'] =  df_train['Physical-Height'] * lbs_to_kg\ndf_train['Physical-Weight'] =  df_train['Physical-Weight'] * inches_to_cm\ndf_train['Physical-Waist_Circumference'] = df_train['Physical-Waist_Circumference'] * inches_to_cm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:41.525339Z","iopub.execute_input":"2024-12-19T21:03:41.525645Z","iopub.status.idle":"2024-12-19T21:03:41.531778Z","shell.execute_reply.started":"2024-12-19T21:03:41.525604Z","shell.execute_reply":"2024-12-19T21:03:41.530782Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Recalculate BMI: BMI = weight (kg) / (height (m)^2) for df_train\ndf_train['Physical-BMI'] = np.where(\n    df_train['Physical-Weight'].notna() & df_train['Physical-Height'].notna(),\n    df_train['Physical-Weight'] / ((df_train['Physical-Height'] / 100) ** 2),\n    np.nan  # If either is NaN, set BMI to NaN\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:41.969811Z","iopub.execute_input":"2024-12-19T21:03:41.970166Z","iopub.status.idle":"2024-12-19T21:03:41.976016Z","shell.execute_reply.started":"2024-12-19T21:03:41.970136Z","shell.execute_reply":"2024-12-19T21:03:41.975134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Transfer units to cm and kg for df_test\nlbs_to_kg = 0.453592\ninches_to_cm = 2.54\n\ndf_test['Physical-Height'] =  df_test['Physical-Height'] * lbs_to_kg\ndf_test['Physical-Weight'] =  df_test['Physical-Weight'] * inches_to_cm\ndf_test['Physical-Waist_Circumference'] = df_test['Physical-Waist_Circumference'] * inches_to_cm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:43.712142Z","iopub.execute_input":"2024-12-19T21:03:43.712445Z","iopub.status.idle":"2024-12-19T21:03:43.718697Z","shell.execute_reply.started":"2024-12-19T21:03:43.712424Z","shell.execute_reply":"2024-12-19T21:03:43.717825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Recalculate BMI: BMI = weight (kg) / (height (m)^2) for df_test\ndf_test['Physical-BMI'] = np.where(\n    df_test['Physical-Weight'].notna() & df_test['Physical-Height'].notna(),\n    df_test['Physical-Weight'] / ((df_test['Physical-Height'] / 100) ** 2),\n    np.nan  # If either is NaN, set BMI to NaN\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:44.171535Z","iopub.execute_input":"2024-12-19T21:03:44.171793Z","iopub.status.idle":"2024-12-19T21:03:44.177523Z","shell.execute_reply.started":"2024-12-19T21:03:44.171772Z","shell.execute_reply":"2024-12-19T21:03:44.176579Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Feature Exteraction","metadata":{}},{"cell_type":"code","source":"# Anthropometric Features df_train\ndf_train['WHR'] = df_train['Physical-Weight'] / (df_train['Physical-Height'] / 100)\ndf_train['Pulse_Pressure'] = df_train['Physical-Systolic_BP'] - df_train['Physical-Diastolic_BP']\ndf_train['BSA'] = 0.007184 * (df_train['Physical-Height'] ** 0.725) * (df_train['Physical-Weight'] ** 0.425)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:44.907265Z","iopub.execute_input":"2024-12-19T21:03:44.907546Z","iopub.status.idle":"2024-12-19T21:03:44.914868Z","shell.execute_reply.started":"2024-12-19T21:03:44.907525Z","shell.execute_reply":"2024-12-19T21:03:44.913937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Bioelectrical Impedance Analysis (BIA) df_train\ndf_train['FFMI'] = df_train['BIA-BIA_FFM'] / ((df_train['Physical-Height'] / 100) ** 2)\ndf_train['ECW_TBW'] = df_train['BIA-BIA_ECW'] / df_train['BIA-BIA_TBW']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:45.397184Z","iopub.execute_input":"2024-12-19T21:03:45.397508Z","iopub.status.idle":"2024-12-19T21:03:45.403130Z","shell.execute_reply.started":"2024-12-19T21:03:45.397483Z","shell.execute_reply":"2024-12-19T21:03:45.402232Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Anthropometric Features df_test\ndf_test['WHR'] = df_test['Physical-Weight'] / (df_test['Physical-Height'] / 100)\ndf_test['Pulse_Pressure'] = df_test['Physical-Systolic_BP'] - df_test['Physical-Diastolic_BP']\ndf_test['BSA'] = 0.007184 * (df_test['Physical-Height'] ** 0.725) * (df_test['Physical-Weight'] ** 0.425)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:46.953154Z","iopub.execute_input":"2024-12-19T21:03:46.953451Z","iopub.status.idle":"2024-12-19T21:03:46.960366Z","shell.execute_reply.started":"2024-12-19T21:03:46.953429Z","shell.execute_reply":"2024-12-19T21:03:46.959501Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Bioelectrical Impedance Analysis (BIA) df_test\ndf_test['FFMI'] = df_test['BIA-BIA_FFM'] / ((df_test['Physical-Height'] / 100) ** 2)\ndf_test['ECW_TBW'] = df_test['BIA-BIA_ECW'] / df_test['BIA-BIA_TBW']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:47.268349Z","iopub.execute_input":"2024-12-19T21:03:47.268640Z","iopub.status.idle":"2024-12-19T21:03:47.274729Z","shell.execute_reply.started":"2024-12-19T21:03:47.268619Z","shell.execute_reply":"2024-12-19T21:03:47.273786Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"In this section, we will eliminate columns where more than half of the values are missing.","metadata":{}},{"cell_type":"code","source":"# Name the column as \"missing\" and count the missing values.\nnull_df = df_train.isna().sum().sort_values(ascending = False).head(60)\nnull_df = pd.DataFrame(null_df)\nnull_df = null_df.rename(columns= {0:'Missing'})\nnull_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:48.293284Z","iopub.execute_input":"2024-12-19T21:03:48.293562Z","iopub.status.idle":"2024-12-19T21:03:48.307415Z","shell.execute_reply.started":"2024-12-19T21:03:48.293540Z","shell.execute_reply":"2024-12-19T21:03:48.306521Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate the number of missing values per column\nmissing_values = df_train.isnull().sum()\n\n# Calculate the threshold for more than half of the values being missing\nthreshold = len(df_train) / 2\n\n# Identify columns with more than half missing values\ncolumns_to_drop = missing_values[missing_values > threshold].index\n\n# Drop the columns from the DataFrame\ndf_train_cleaned = df_train.drop(columns=columns_to_drop)\n\nprint(\"Columns dropped:\", columns_to_drop)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:48.745423Z","iopub.execute_input":"2024-12-19T21:03:48.745754Z","iopub.status.idle":"2024-12-19T21:03:48.757916Z","shell.execute_reply.started":"2024-12-19T21:03:48.745726Z","shell.execute_reply":"2024-12-19T21:03:48.757079Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df_train['sii'] = y\ndf_train = pd.DataFrame(df_train)\ndf_train = pd.concat([df_train, y], axis=1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:51.599151Z","iopub.execute_input":"2024-12-19T21:03:51.599467Z","iopub.status.idle":"2024-12-19T21:03:51.608984Z","shell.execute_reply.started":"2024-12-19T21:03:51.599444Z","shell.execute_reply":"2024-12-19T21:03:51.608131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:52.149846Z","iopub.execute_input":"2024-12-19T21:03:52.150145Z","iopub.status.idle":"2024-12-19T21:03:52.186174Z","shell.execute_reply.started":"2024-12-19T21:03:52.150122Z","shell.execute_reply":"2024-12-19T21:03:52.185378Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Convert categorical columns to numeric using LabelEncoder for both train and test dataset","metadata":{}},{"cell_type":"code","source":"# df_train\n# df_train = pd.DataFrame(df_train)\nfrom sklearn.preprocessing import LabelEncoder\n# Label ecoder on churn and contract\nlabel_col = df_train\n\n# Convert categorical columns to numeric using LabelEncoder\nlabel_encoder = {}\nfor column in label_col :\n    label = LabelEncoder()\n    df_train[column] = label.fit_transform(df_train[column])\n    label_encoder[column] = label\ndf_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:54.355197Z","iopub.execute_input":"2024-12-19T21:03:54.355543Z","iopub.status.idle":"2024-12-19T21:03:54.411677Z","shell.execute_reply.started":"2024-12-19T21:03:54.355516Z","shell.execute_reply":"2024-12-19T21:03:54.410779Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df_test\nfrom sklearn.preprocessing import LabelEncoder\n# Label ecoder on churn and contract\nlabel_col = df_test\n\n# Convert categorical columns to numeric using LabelEncoder\nlabel_encoder = {}\nfor column in label_col :\n    label = LabelEncoder()\n    df_test[column] = label.fit_transform(df_test[column])\n    label_encoder[column] = label\ndf_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:03:56.097419Z","iopub.execute_input":"2024-12-19T21:03:56.097748Z","iopub.status.idle":"2024-12-19T21:03:56.139423Z","shell.execute_reply.started":"2024-12-19T21:03:56.097720Z","shell.execute_reply":"2024-12-19T21:03:56.138750Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"#  corr between the faetures","metadata":{}},{"cell_type":"code","source":"# visualizing the corr between the faetures\ncorr_matrix = df_train.corr()\n\nplt.figure(figsize = (20, 20))\nsns.heatmap(corr_matrix, annot = True, fmt = '.2f', cmap = 'coolwarm', cbar = True)\nplt.title(\"Correlation Matrix Heatmap\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T23:44:32.047937Z","iopub.execute_input":"2024-11-29T23:44:32.048268Z","iopub.status.idle":"2024-11-29T23:44:39.405499Z","shell.execute_reply.started":"2024-11-29T23:44:32.048240Z","shell.execute_reply":"2024-11-29T23:44:39.404088Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# features that are highly correlated with the target variable\ncorr_with_target = corr_matrix['sii'].sort_values(ascending = False)\nprint(corr_with_target)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T23:44:39.407367Z","iopub.execute_input":"2024-11-29T23:44:39.408006Z","iopub.status.idle":"2024-11-29T23:44:39.414728Z","shell.execute_reply.started":"2024-11-29T23:44:39.407966Z","shell.execute_reply":"2024-11-29T23:44:39.413992Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# feature importances","metadata":{}},{"cell_type":"code","source":"X = df_train.drop('sii', axis = 1)\ny = df_train['sii']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T23:44:44.163111Z","iopub.execute_input":"2024-11-29T23:44:44.163774Z","iopub.status.idle":"2024-11-29T23:44:44.171433Z","shell.execute_reply.started":"2024-11-29T23:44:44.163743Z","shell.execute_reply":"2024-11-29T23:44:44.170675Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = RandomForestClassifier(random_state = 42)\nmodel.fit(X, y)\n\n# Get feature importances\nimportances = model.feature_importances_\n\n# Create a DataFrame for better visualization\nfeature_importances = pd.DataFrame({\n    'Feature' : X.columns,\n    'Importances' : importances}).sort_values(by = 'Importances', ascending = False).head(30)\n\n# Visualize the most important features\nplt.figure(figsize = (10, 6))\nsns.barplot(x = 'Importances', y = 'Feature', data = feature_importances)\nplt.title(\"Feature Importances\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T23:44:51.462975Z","iopub.execute_input":"2024-11-29T23:44:51.463598Z","iopub.status.idle":"2024-11-29T23:44:52.889799Z","shell.execute_reply.started":"2024-11-29T23:44:51.463566Z","shell.execute_reply":"2024-11-29T23:44:52.888942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# df_train = df_train[selected_features]\n# df_test = df_test[selected_features]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-29T22:03:23.288242Z","iopub.execute_input":"2024-11-29T22:03:23.288537Z","iopub.status.idle":"2024-11-29T22:03:23.295802Z","shell.execute_reply.started":"2024-11-29T22:03:23.288512Z","shell.execute_reply":"2024-11-29T22:03:23.294915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-17T23:51:17.620077Z","iopub.execute_input":"2024-12-17T23:51:17.620904Z","iopub.status.idle":"2024-12-17T23:51:17.626352Z","shell.execute_reply.started":"2024-12-17T23:51:17.620869Z","shell.execute_reply":"2024-12-17T23:51:17.625525Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# y_train\ny_train = df_train['sii']\ndf_train = df_train.drop(['sii'], axis = 1)\n# df_test = df_test.drop(['id'], axis = 1)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:04:01.721953Z","iopub.execute_input":"2024-12-19T21:04:01.722297Z","iopub.status.idle":"2024-12-19T21:04:01.729269Z","shell.execute_reply.started":"2024-12-19T21:04:01.722270Z","shell.execute_reply":"2024-12-19T21:04:01.728402Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Simple Imputer for missing data ","metadata":{}},{"cell_type":"code","source":"imputer = SimpleImputer(strategy = 'mean')\ndf_train = imputer.fit_transform(df_train)\ndf_test = pd.DataFrame(imputer.transform(df_test), columns = df_test.columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:04:03.251417Z","iopub.execute_input":"2024-12-19T21:04:03.251692Z","iopub.status.idle":"2024-12-19T21:04:03.270175Z","shell.execute_reply.started":"2024-12-19T21:04:03.251670Z","shell.execute_reply":"2024-12-19T21:04:03.269472Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# split the data\nX_train = df_train\nX_test = df_test","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:04:04.675325Z","iopub.execute_input":"2024-12-19T21:04:04.675608Z","iopub.status.idle":"2024-12-19T21:04:04.679550Z","shell.execute_reply.started":"2024-12-19T21:04:04.675587Z","shell.execute_reply":"2024-12-19T21:04:04.678571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Normalize the data\nscaler = StandardScaler()\nX_train = pd.DataFrame(scaler.fit_transform(X_train))\nX_test = pd.DataFrame(scaler.transform(X_test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:04:05.074154Z","iopub.execute_input":"2024-12-19T21:04:05.074464Z","iopub.status.idle":"2024-12-19T21:04:05.086585Z","shell.execute_reply.started":"2024-12-19T21:04:05.074437Z","shell.execute_reply":"2024-12-19T21:04:05.085728Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# SmoteNN for oversampling","metadata":{}},{"cell_type":"code","source":"# Apply SMOTEENN for both oversampling and cleaning\nfrom imblearn.combine import SMOTEENN\nsmote_enn = SMOTEENN(random_state=42)\nX_res, y_res = smote_enn.fit_resample(X_train, y_train)\n\ndf = pd.concat([X_res, y_res], axis=1)\n\nprint(pd.Series(y_res).value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:04:11.035271Z","iopub.execute_input":"2024-12-19T21:04:11.035605Z","iopub.status.idle":"2024-12-19T21:04:11.463161Z","shell.execute_reply.started":"2024-12-19T21:04:11.035577Z","shell.execute_reply":"2024-12-19T21:04:11.462323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# X_train_split, X_val, y_train_split, y_val = train_test_split(X_res, y_res, test_size=0.1, random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-09T19:26:55.020023Z","iopub.execute_input":"2024-12-09T19:26:55.021051Z","iopub.status.idle":"2024-12-09T19:26:55.024593Z","shell.execute_reply.started":"2024-12-09T19:26:55.021013Z","shell.execute_reply":"2024-12-09T19:26:55.023707Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PCA","metadata":{}},{"cell_type":"code","source":"# PCA for dimensionality reduction\nfrom sklearn.decomposition import PCA\n\npca = PCA(n_components=0.90)\nX_pca = pca.fit_transform(X_res)\n\nX_test_pca = pca.transform(X_test)\nprint(f\"Number of components selected X_train_pca: {pca.n_components_}\")\nprint(f\"Number of components selected X_test_pca: {X_test_pca.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:04:30.777767Z","iopub.execute_input":"2024-12-19T21:04:30.778232Z","iopub.status.idle":"2024-12-19T21:04:30.827664Z","shell.execute_reply.started":"2024-12-19T21:04:30.778201Z","shell.execute_reply":"2024-12-19T21:04:30.826227Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Set the test size to 0.05 to minimize the amount of data allocated for testing, ensuring the majority remains available for training.","metadata":{}},{"cell_type":"code","source":"# Split the resampled dataset\nX_train, X_val, y_train, y_val = train_test_split(X_pca, y_res, test_size = 0.05, random_state = 42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:04:32.114233Z","iopub.execute_input":"2024-12-19T21:04:32.114516Z","iopub.status.idle":"2024-12-19T21:04:32.120294Z","shell.execute_reply.started":"2024-12-19T21:04:32.114495Z","shell.execute_reply":"2024-12-19T21:04:32.119542Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train the Model","metadata":{}},{"cell_type":"code","source":"import xgboost as xgb\n\n# Initialize an XGBoost classifier\nxgb_model = xgb.XGBClassifier(use_label_encoder=False, eval_metric='logloss', random_state=42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:04:34.830970Z","iopub.execute_input":"2024-12-19T21:04:34.831311Z","iopub.status.idle":"2024-12-19T21:04:35.021690Z","shell.execute_reply.started":"2024-12-19T21:04:34.831284Z","shell.execute_reply":"2024-12-19T21:04:35.021078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the parameter grid\nparam_grid = {\n    'n_estimators': [50, 100, 200],\n    'learning_rate': [0.01, 0.1, 0.2],\n    'max_depth': [3, 5, 7],\n    'subsample': [0.8, 1],\n    'colsample_bytree': [0.8, 1],\n    'gamma': [0, 1, 5]\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:04:36.334703Z","iopub.execute_input":"2024-12-19T21:04:36.335057Z","iopub.status.idle":"2024-12-19T21:04:36.339765Z","shell.execute_reply.started":"2024-12-19T21:04:36.334991Z","shell.execute_reply":"2024-12-19T21:04:36.338654Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set up the GridSearchCV\ngrid_search_xgb = GridSearchCV(\n    estimator=xgb_model,\n    param_grid=param_grid,\n    scoring='accuracy',\n    cv=5,\n    verbose=2,\n    n_jobs=-1,  # Use all available cores\n#     callbacks=[early_stopping]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:04:37.339633Z","iopub.execute_input":"2024-12-19T21:04:37.339953Z","iopub.status.idle":"2024-12-19T21:04:37.344185Z","shell.execute_reply.started":"2024-12-19T21:04:37.339927Z","shell.execute_reply":"2024-12-19T21:04:37.343305Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# Fit the model\ngrid_search_xgb.fit(X_train, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:04:39.978496Z","iopub.execute_input":"2024-12-19T21:04:39.978776Z","iopub.status.idle":"2024-12-19T21:16:12.771225Z","shell.execute_reply.started":"2024-12-19T21:04:39.978753Z","shell.execute_reply":"2024-12-19T21:16:12.770178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n# Save the trained model using pickle\nwith open('/kaggle/working/trained_model.pkl', 'wb') as file:\n    pickle.dump(grid_search_xgb, file)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:17:10.328140Z","iopub.execute_input":"2024-12-19T21:17:10.328487Z","iopub.status.idle":"2024-12-19T21:17:10.347773Z","shell.execute_reply.started":"2024-12-19T21:17:10.328457Z","shell.execute_reply":"2024-12-19T21:17:10.347156Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print best parameters\nprint(\"\\nBest Hyperparameters:\", grid_search_xgb.best_params_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:17:13.100591Z","iopub.execute_input":"2024-12-19T21:17:13.100897Z","iopub.status.idle":"2024-12-19T21:17:13.105738Z","shell.execute_reply.started":"2024-12-19T21:17:13.100873Z","shell.execute_reply":"2024-12-19T21:17:13.104880Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate Model on Validation Data\nbest_model_xgb = grid_search_xgb.best_estimator_\nscore = best_model_xgb.score(X_train, y_train)\ny_pred = best_model_xgb.predict(X_val)\nprint(score)\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_val, y_pred))\nprint(\"\\nConfusion Matrix:\")\nprint(confusion_matrix(y_val, y_pred))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:17:14.823397Z","iopub.execute_input":"2024-12-19T21:17:14.823678Z","iopub.status.idle":"2024-12-19T21:17:14.898203Z","shell.execute_reply.started":"2024-12-19T21:17:14.823658Z","shell.execute_reply":"2024-12-19T21:17:14.897349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print Overall Accuracy on Validation\naccuracy = accuracy_score(y_val, y_pred)\nprint(f\"\\nValidation Accuracy: {accuracy:}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:17:18.350683Z","iopub.execute_input":"2024-12-19T21:17:18.350991Z","iopub.status.idle":"2024-12-19T21:17:18.357740Z","shell.execute_reply.started":"2024-12-19T21:17:18.350970Z","shell.execute_reply":"2024-12-19T21:17:18.356842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on Test Data\ntest_predictions = best_model_xgb.predict(X_test_pca)\nprint(\"\\nTest Predictions:\", test_predictions[:10])  # Display first 10 predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:17:19.900903Z","iopub.execute_input":"2024-12-19T21:17:19.901337Z","iopub.status.idle":"2024-12-19T21:17:19.909143Z","shell.execute_reply.started":"2024-12-19T21:17:19.901306Z","shell.execute_reply":"2024-12-19T21:17:19.908099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# Prepare the submission DataFrame\nsubmission = pd.DataFrame({\n    'id': df_sub_samp['id'],  # Assuming 'Id' column exists in the test data\n    'Prediction': test_predictions\n})\n\n# Save the submission file\nsubmission.to_csv('submission.csv', index=False)\n\nprint(\"Submission file saved as 'submission.csv'.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:17:21.535979Z","iopub.execute_input":"2024-12-19T21:17:21.536348Z","iopub.status.idle":"2024-12-19T21:17:21.545874Z","shell.execute_reply.started":"2024-12-19T21:17:21.536311Z","shell.execute_reply":"2024-12-19T21:17:21.544959Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-19T21:17:23.683117Z","iopub.execute_input":"2024-12-19T21:17:23.683415Z","iopub.status.idle":"2024-12-19T21:17:23.692026Z","shell.execute_reply.started":"2024-12-19T21:17:23.683394Z","shell.execute_reply":"2024-12-19T21:17:23.691127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}