{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Harmful Brain Activity Classification - Boosting methods // ML**\n\n> -**XGBoost**\n\n> -**Random Forest Classifier**\n\n> -**Decision Tree Classifier**\n\n## **Written by:** [Aarish Asif Khan](https://www.kaggle.com/aarishasifkhan)\n\n## **Date:** 18th February 2024\n\n## **Dataset:** [HMC - Harmful Brain Activity Dataset](https://www.kaggle.com/competitions/hms-harmful-brain-activity-classification)","metadata":{}},{"cell_type":"markdown","source":"1. # **First Model - Decision Tree Classifier**","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport pandas as pd \nimport numpy as np \nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:22:53.795301Z","iopub.execute_input":"2024-03-18T12:22:53.795738Z","iopub.status.idle":"2024-03-18T12:22:57.931764Z","shell.execute_reply.started":"2024-03-18T12:22:53.795703Z","shell.execute_reply":"2024-03-18T12:22:57.930124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the HMC dataset\ntrain_data = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\n\n# Print the 5 rows of the dataset\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:22:57.934121Z","iopub.execute_input":"2024-03-18T12:22:57.934803Z","iopub.status.idle":"2024-03-18T12:22:58.279405Z","shell.execute_reply.started":"2024-03-18T12:22:57.934756Z","shell.execute_reply":"2024-03-18T12:22:58.277924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign values to X and y\nX = train_data.drop(['eeg_id', 'eeg_sub_id', 'eeg_label_offset_seconds', 'spectrogram_id', 'spectrogram_sub_id', 'spectrogram_label_offset_seconds', 'label_id', 'patient_id', 'expert_consensus', 'seizure_vote'], axis=1)\ny = train_data['expert_consensus']","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:22:58.281095Z","iopub.execute_input":"2024-03-18T12:22:58.281900Z","iopub.status.idle":"2024-03-18T12:22:58.295964Z","shell.execute_reply.started":"2024-03-18T12:22:58.281850Z","shell.execute_reply":"2024-03-18T12:22:58.294204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train-test-split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:22:58.300082Z","iopub.execute_input":"2024-03-18T12:22:58.300724Z","iopub.status.idle":"2024-03-18T12:22:58.332413Z","shell.execute_reply.started":"2024-03-18T12:22:58.300674Z","shell.execute_reply":"2024-03-18T12:22:58.330793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build and Train the model\ndtc = DecisionTreeClassifier(random_state=42)\ndtc.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:22:58.334320Z","iopub.execute_input":"2024-03-18T12:22:58.334977Z","iopub.status.idle":"2024-03-18T12:22:58.715487Z","shell.execute_reply.started":"2024-03-18T12:22:58.334927Z","shell.execute_reply":"2024-03-18T12:22:58.714456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict the model\ny_pred = dtc.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:22:58.716937Z","iopub.execute_input":"2024-03-18T12:22:58.718229Z","iopub.status.idle":"2024-03-18T12:22:58.727078Z","shell.execute_reply.started":"2024-03-18T12:22:58.718178Z","shell.execute_reply":"2024-03-18T12:22:58.725909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\nprint('Decision Tree Classifier:')\nprint('Accuracy:', accuracy_score(y_test, y_pred))\nprint('Precision:', precision_score(y_test, y_pred, average='weighted'))\nprint('Recall:', recall_score(y_test, y_pred, average='weighted'))\nprint('F1 Score:', f1_score(y_test, y_pred, average='weighted'))","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:22:58.728839Z","iopub.execute_input":"2024-03-18T12:22:58.729261Z","iopub.status.idle":"2024-03-18T12:22:59.834038Z","shell.execute_reply.started":"2024-03-18T12:22:58.729220Z","shell.execute_reply":"2024-03-18T12:22:59.832687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Classification Report\nprint(\"Classification Report (Decision Tree):\")\nprint(classification_report(y_test, y_pred))\n","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:22:59.835941Z","iopub.execute_input":"2024-03-18T12:22:59.836745Z","iopub.status.idle":"2024-03-18T12:23:01.242300Z","shell.execute_reply.started":"2024-03-18T12:22:59.836686Z","shell.execute_reply":"2024-03-18T12:23:01.240867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2. # **Second Model - Random Forest Classifier**","metadata":{}},{"cell_type":"code","source":"# Import libraries\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:01.244155Z","iopub.execute_input":"2024-03-18T12:23:01.244770Z","iopub.status.idle":"2024-03-18T12:23:01.330895Z","shell.execute_reply.started":"2024-03-18T12:23:01.244733Z","shell.execute_reply":"2024-03-18T12:23:01.329725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build and Train the model\nrfc = RandomForestClassifier(n_estimators=100, random_state=42)\nrfc.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:01.335682Z","iopub.execute_input":"2024-03-18T12:23:01.336086Z","iopub.status.idle":"2024-03-18T12:23:06.764677Z","shell.execute_reply.started":"2024-03-18T12:23:01.336056Z","shell.execute_reply":"2024-03-18T12:23:06.763539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict the model\ny_pred = rfc.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:06.766122Z","iopub.execute_input":"2024-03-18T12:23:06.766762Z","iopub.status.idle":"2024-03-18T12:23:07.057956Z","shell.execute_reply.started":"2024-03-18T12:23:06.766732Z","shell.execute_reply":"2024-03-18T12:23:07.056668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\nprint('Random Forest Classifier:')\nprint('Accuracy:', accuracy_score(y_test, y_pred))\nprint('Precision:', precision_score(y_test, y_pred, average='weighted'))\nprint('Recall:', recall_score(y_test, y_pred, average='weighted'))\nprint('F1 Score:', f1_score(y_test, y_pred, average='weighted'))","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:07.059223Z","iopub.execute_input":"2024-03-18T12:23:07.059566Z","iopub.status.idle":"2024-03-18T12:23:08.069415Z","shell.execute_reply.started":"2024-03-18T12:23:07.059536Z","shell.execute_reply":"2024-03-18T12:23:08.068041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Classification Report\nprint(\"Classification Report (Random Forest):\")\nprint(classification_report(y_test, y_pred))\n","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:08.070501Z","iopub.execute_input":"2024-03-18T12:23:08.071126Z","iopub.status.idle":"2024-03-18T12:23:09.477801Z","shell.execute_reply.started":"2024-03-18T12:23:08.071094Z","shell.execute_reply":"2024-03-18T12:23:09.476443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"3. # **Third Model - XGBoost**","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import accuracy_score, precision_score\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:09.479248Z","iopub.execute_input":"2024-03-18T12:23:09.479723Z","iopub.status.idle":"2024-03-18T12:23:09.684252Z","shell.execute_reply.started":"2024-03-18T12:23:09.479618Z","shell.execute_reply":"2024-03-18T12:23:09.683361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the dataset\ntrain_data = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\n\n# Display the first few rows of the dataset to understand its structure\nprint(train_data.head())","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:09.685448Z","iopub.execute_input":"2024-03-18T12:23:09.686469Z","iopub.status.idle":"2024-03-18T12:23:09.927956Z","shell.execute_reply.started":"2024-03-18T12:23:09.686431Z","shell.execute_reply":"2024-03-18T12:23:09.926647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define features and target variable\nfeatures = train_data.drop(['eeg_id', 'eeg_sub_id', 'eeg_label_offset_seconds', 'spectrogram_id', \n            'spectrogram_sub_id', 'spectrogram_label_offset_seconds', 'patient_id', 'expert_consensus', 'seizure_vote'], axis=1)\n\ntarget = 'seizure_vote'","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:09.929479Z","iopub.execute_input":"2024-03-18T12:23:09.929945Z","iopub.status.idle":"2024-03-18T12:23:09.936641Z","shell.execute_reply.started":"2024-03-18T12:23:09.929899Z","shell.execute_reply":"2024-03-18T12:23:09.935677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove unnecessary columns\ntrain_data = train_data.drop(['label_id'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:09.937765Z","iopub.execute_input":"2024-03-18T12:23:09.938697Z","iopub.status.idle":"2024-03-18T12:23:09.954627Z","shell.execute_reply.started":"2024-03-18T12:23:09.938662Z","shell.execute_reply":"2024-03-18T12:23:09.953023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert categorical variables to numerical using one-hot encoding\ntrain_data = pd.get_dummies(train_data)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:09.956050Z","iopub.execute_input":"2024-03-18T12:23:09.956452Z","iopub.status.idle":"2024-03-18T12:23:09.990185Z","shell.execute_reply.started":"2024-03-18T12:23:09.956419Z","shell.execute_reply":"2024-03-18T12:23:09.988673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split data into features and target variable\nX = train_data.drop(columns=[target], axis=1)\ny = train_data[target]","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:09.991670Z","iopub.execute_input":"2024-03-18T12:23:09.992048Z","iopub.status.idle":"2024-03-18T12:23:10.001463Z","shell.execute_reply.started":"2024-03-18T12:23:09.992017Z","shell.execute_reply":"2024-03-18T12:23:10.000001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the data into training and testing sets (80% training, 20% testing)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:10.003058Z","iopub.execute_input":"2024-03-18T12:23:10.003476Z","iopub.status.idle":"2024-03-18T12:23:10.030834Z","shell.execute_reply.started":"2024-03-18T12:23:10.003422Z","shell.execute_reply":"2024-03-18T12:23:10.029466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize LabelEncoder\nlabel_encoder = LabelEncoder()","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:10.032597Z","iopub.execute_input":"2024-03-18T12:23:10.033375Z","iopub.status.idle":"2024-03-18T12:23:10.039365Z","shell.execute_reply.started":"2024-03-18T12:23:10.033328Z","shell.execute_reply":"2024-03-18T12:23:10.037840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fit LabelEncoder on the combined target variable (y_train and y_test) to ensure consistent encoding\nlabel_encoder.fit(pd.concat([y_train, y_test]))","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:10.041044Z","iopub.execute_input":"2024-03-18T12:23:10.041509Z","iopub.status.idle":"2024-03-18T12:23:10.061948Z","shell.execute_reply.started":"2024-03-18T12:23:10.041462Z","shell.execute_reply":"2024-03-18T12:23:10.060946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transform target variables to encoded labels\ny_train_encoded = label_encoder.transform(y_train)\ny_test_encoded = label_encoder.transform(y_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:10.063426Z","iopub.execute_input":"2024-03-18T12:23:10.063835Z","iopub.status.idle":"2024-03-18T12:23:10.074646Z","shell.execute_reply.started":"2024-03-18T12:23:10.063798Z","shell.execute_reply":"2024-03-18T12:23:10.073357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize the XGBoost classifier\nmodel = XGBClassifier()\n\n# Train the model\nmodel.fit(X_train, y_train_encoded)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:10.077162Z","iopub.execute_input":"2024-03-18T12:23:10.078209Z","iopub.status.idle":"2024-03-18T12:23:22.776696Z","shell.execute_reply.started":"2024-03-18T12:23:10.078149Z","shell.execute_reply":"2024-03-18T12:23:22.775443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predict the model\ny_pred = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:22.778490Z","iopub.execute_input":"2024-03-18T12:23:22.779288Z","iopub.status.idle":"2024-03-18T12:23:23.149573Z","shell.execute_reply.started":"2024-03-18T12:23:22.779238Z","shell.execute_reply":"2024-03-18T12:23:23.148583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\nprint('XGBoost:')\nprint('Accuracy:', accuracy_score(y_test_encoded, y_pred))\nprint('Precision:', precision_score(y_test_encoded, y_pred, average='weighted'))\nprint('Recall:', recall_score(y_test_encoded, y_pred, average='weighted'))\nprint('F1 Score:', f1_score(y_test_encoded, y_pred, average='weighted'))","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:23.151345Z","iopub.execute_input":"2024-03-18T12:23:23.152239Z","iopub.status.idle":"2024-03-18T12:23:23.201721Z","shell.execute_reply.started":"2024-03-18T12:23:23.152193Z","shell.execute_reply":"2024-03-18T12:23:23.200421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, classification_report\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Classification Report\nprint(\"Classification Report:\")\nprint(classification_report(y_test_encoded, y_pred))\n","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:23:23.205762Z","iopub.execute_input":"2024-03-18T12:23:23.207085Z","iopub.status.idle":"2024-03-18T12:23:23.258836Z","shell.execute_reply.started":"2024-03-18T12:23:23.207043Z","shell.execute_reply":"2024-03-18T12:23:23.257854Z"},"trusted":true},"execution_count":null,"outputs":[]}]}