{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Harmful Brain Activity Classification - Random Forest Algorithm**\n\n## **Written by:** Aarish Asif Khan\n\n## **Date:** 18th February 2024","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport pandas as pd\nimport numpy as np \nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\n\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:24.932615Z","iopub.execute_input":"2024-03-10T16:36:24.933255Z","iopub.status.idle":"2024-03-10T16:36:24.941709Z","shell.execute_reply.started":"2024-03-10T16:36:24.933199Z","shell.execute_reply":"2024-03-10T16:36:24.940727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the dataset\ntrain_data = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:24.943576Z","iopub.execute_input":"2024-03-10T16:36:24.944155Z","iopub.status.idle":"2024-03-10T16:36:25.192582Z","shell.execute_reply.started":"2024-03-10T16:36:24.944120Z","shell.execute_reply":"2024-03-10T16:36:25.191269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop unnecessary columns\ntrain_data.drop(['eeg_id', 'eeg_sub_id', 'spectrogram_id', 'spectrogram_sub_id', 'label_id', 'patient_id'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:25.194035Z","iopub.execute_input":"2024-03-10T16:36:25.194471Z","iopub.status.idle":"2024-03-10T16:36:25.204778Z","shell.execute_reply.started":"2024-03-10T16:36:25.194436Z","shell.execute_reply":"2024-03-10T16:36:25.203485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encode categorical variables\nlabel_encoder = LabelEncoder()\ntrain_data['expert_consensus'] = label_encoder.fit_transform(train_data['expert_consensus'])","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:25.207658Z","iopub.execute_input":"2024-03-10T16:36:25.208034Z","iopub.status.idle":"2024-03-10T16:36:25.248812Z","shell.execute_reply.started":"2024-03-10T16:36:25.208005Z","shell.execute_reply":"2024-03-10T16:36:25.247203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split features and target variable\nX = train_data.drop('expert_consensus', axis=1)\ny = train_data['expert_consensus']","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:25.250438Z","iopub.execute_input":"2024-03-10T16:36:25.251895Z","iopub.status.idle":"2024-03-10T16:36:25.265647Z","shell.execute_reply.started":"2024-03-10T16:36:25.251857Z","shell.execute_reply":"2024-03-10T16:36:25.264421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:25.267478Z","iopub.execute_input":"2024-03-10T16:36:25.268252Z","iopub.status.idle":"2024-03-10T16:36:25.293794Z","shell.execute_reply.started":"2024-03-10T16:36:25.268214Z","shell.execute_reply":"2024-03-10T16:36:25.292258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the random forest classifier\nrf_classifier = RandomForestClassifier(random_state=42)\nrf_classifier.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:25.295392Z","iopub.execute_input":"2024-03-10T16:36:25.295797Z","iopub.status.idle":"2024-03-10T16:36:33.146479Z","shell.execute_reply.started":"2024-03-10T16:36:25.295764Z","shell.execute_reply":"2024-03-10T16:36:33.145155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions\ny_pred = rf_classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:33.148364Z","iopub.execute_input":"2024-03-10T16:36:33.148768Z","iopub.status.idle":"2024-03-10T16:36:33.470423Z","shell.execute_reply.started":"2024-03-10T16:36:33.148733Z","shell.execute_reply":"2024-03-10T16:36:33.469136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the classifier\naccuracy = accuracy_score(y_test, y_pred)\nprint(\"Accuracy:\", accuracy)\nprint(\"Classification Report:\\n\", classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:33.473570Z","iopub.execute_input":"2024-03-10T16:36:33.474395Z","iopub.status.idle":"2024-03-10T16:36:33.541146Z","shell.execute_reply.started":"2024-03-10T16:36:33.474349Z","shell.execute_reply":"2024-03-10T16:36:33.539817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_importances = pd.Series(rf_classifier.feature_importances_, index=X.columns)\nfeat_importances.nlargest(10).plot(kind='barh', color=['blue', 'green', 'red', 'purple', 'lightblue', 'yellow', 'cyan', 'magenta', 'gray', 'pink'])\nplt.title('Top 10 Most Important Features')\nplt.xlabel('Feature Importance Score')\nplt.ylabel('Features')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:33.545217Z","iopub.execute_input":"2024-03-10T16:36:33.546074Z","iopub.status.idle":"2024-03-10T16:36:33.915129Z","shell.execute_reply.started":"2024-03-10T16:36:33.546006Z","shell.execute_reply":"2024-03-10T16:36:33.914157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Confusion matrix\nconf_matrix = confusion_matrix(y_test, y_pred)\nplt.figure(figsize=(8, 6))\nsns.heatmap(conf_matrix, annot=True, fmt=\"d\", cmap=\"Blues\", cbar=False, \n            xticklabels=label_encoder.classes_, yticklabels=label_encoder.classes_)\nplt.xlabel('Predicted Labels')\nplt.ylabel('True Labels')\nplt.title('Confusion Matrix')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:36:33.916606Z","iopub.execute_input":"2024-03-10T16:36:33.917197Z","iopub.status.idle":"2024-03-10T16:36:34.210230Z","shell.execute_reply.started":"2024-03-10T16:36:33.917163Z","shell.execute_reply":"2024-03-10T16:36:34.208822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **What this confusion matrix shows:**","metadata":{}},{"cell_type":"markdown","source":"The confusion matrix visually represents the performance of a classification model by comparing the predicted labels with the actual true labels across different classes. Each row of the matrix represents the instances in an actual class, while each column represents the instances in a predicted class.","metadata":{}}]}