{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Harmful Brain Activity Classification - Random Forest, Decision Tree, SVM // EDA (Exploratory Data Analysis)**\n\n## **Author:** [Aarish Asif Khan](https://www.kaggle.com/aarishasifkhan)\n\n## - **`Twitter:`** [@Aarish47](https://twitter.com/Aarish47)\n\n## - **`Github:`** [@aarish47](https://github.com/aarish47)\n\n## **Dataset:** [HMC Dataset](https://www.kaggle.com/competitions/hms-harmful-brain-activity-classification)","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport pandas as pd \nimport numpy as np \n\nimport matplotlib.pyplot as plt \nimport seaborn as sns \n\nimport tensorflow as tf \n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\n\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\n\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\n\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import accuracy_score\n","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:03:59.545290Z","iopub.execute_input":"2024-03-12T17:03:59.545752Z","iopub.status.idle":"2024-03-12T17:04:15.891736Z","shell.execute_reply.started":"2024-03-12T17:03:59.545694Z","shell.execute_reply":"2024-03-12T17:04:15.890331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the `train.csv` dataset\ntrain_data = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:15.893654Z","iopub.execute_input":"2024-03-12T17:04:15.894240Z","iopub.status.idle":"2024-03-12T17:04:16.071148Z","shell.execute_reply.started":"2024-03-12T17:04:15.894211Z","shell.execute_reply":"2024-03-12T17:04:16.069446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print the 5 rows of the dataset\ntrain_data.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.072615Z","iopub.execute_input":"2024-03-12T17:04:16.073020Z","iopub.status.idle":"2024-03-12T17:04:16.098310Z","shell.execute_reply.started":"2024-03-12T17:04:16.072986Z","shell.execute_reply":"2024-03-12T17:04:16.097321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove all the warnings\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.100301Z","iopub.execute_input":"2024-03-12T17:04:16.100617Z","iopub.status.idle":"2024-03-12T17:04:16.105470Z","shell.execute_reply.started":"2024-03-12T17:04:16.100583Z","shell.execute_reply":"2024-03-12T17:04:16.104330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Check for Missing values and Duplicated columns**","metadata":{}},{"cell_type":"code","source":"# Print the dataset information\ntrain_data.info()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.106670Z","iopub.execute_input":"2024-03-12T17:04:16.107100Z","iopub.status.idle":"2024-03-12T17:04:16.153391Z","shell.execute_reply.started":"2024-03-12T17:04:16.107073Z","shell.execute_reply":"2024-03-12T17:04:16.152267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the missing values \ntrain_data.isna().sum().plot(kind='bar')\nplt.title('Missing values in the dataset')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.154726Z","iopub.execute_input":"2024-03-12T17:04:16.155207Z","iopub.status.idle":"2024-03-12T17:04:16.466570Z","shell.execute_reply.started":"2024-03-12T17:04:16.155178Z","shell.execute_reply":"2024-03-12T17:04:16.465299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Since there arent any Missing values, lets check if any columns are Duplicated or not.**","metadata":{}},{"cell_type":"code","source":"# Check for duplicated columns\ntrain_data.duplicated().sum()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.467654Z","iopub.execute_input":"2024-03-12T17:04:16.467953Z","iopub.status.idle":"2024-03-12T17:04:16.507244Z","shell.execute_reply.started":"2024-03-12T17:04:16.467927Z","shell.execute_reply":"2024-03-12T17:04:16.506130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **There arent any duplicate columns**","metadata":{}},{"cell_type":"code","source":"# Shape of the dataset\ntrain_data.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.508596Z","iopub.execute_input":"2024-03-12T17:04:16.508873Z","iopub.status.idle":"2024-03-12T17:04:16.517826Z","shell.execute_reply.started":"2024-03-12T17:04:16.508852Z","shell.execute_reply":"2024-03-12T17:04:16.516361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.519586Z","iopub.execute_input":"2024-03-12T17:04:16.520542Z","iopub.status.idle":"2024-03-12T17:04:16.589305Z","shell.execute_reply.started":"2024-03-12T17:04:16.520500Z","shell.execute_reply":"2024-03-12T17:04:16.587700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check unique values of `expert_consensus`\ntrain_data['expert_consensus'].unique()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.593040Z","iopub.execute_input":"2024-03-12T17:04:16.593409Z","iopub.status.idle":"2024-03-12T17:04:16.605838Z","shell.execute_reply.started":"2024-03-12T17:04:16.593338Z","shell.execute_reply":"2024-03-12T17:04:16.604968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Plotting**","metadata":{}},{"cell_type":"code","source":"# Plot the distribution of `expert_consensus`\nsns.countplot(x='expert_consensus', data=train_data)\nplt.title('Distribution of expert consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.607264Z","iopub.execute_input":"2024-03-12T17:04:16.607843Z","iopub.status.idle":"2024-03-12T17:04:16.874720Z","shell.execute_reply.started":"2024-03-12T17:04:16.607816Z","shell.execute_reply":"2024-03-12T17:04:16.873569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Apply LabelEncoder()**","metadata":{}},{"cell_type":"code","source":"# Encode the `expert_consensus` column\nlabel_encoder = LabelEncoder()\ntrain_data['expert_consensus'] = label_encoder.fit_transform(train_data['expert_consensus'])","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.876212Z","iopub.execute_input":"2024-03-12T17:04:16.876589Z","iopub.status.idle":"2024-03-12T17:04:16.901581Z","shell.execute_reply.started":"2024-03-12T17:04:16.876560Z","shell.execute_reply":"2024-03-12T17:04:16.900472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.903029Z","iopub.execute_input":"2024-03-12T17:04:16.903324Z","iopub.status.idle":"2024-03-12T17:04:16.927675Z","shell.execute_reply.started":"2024-03-12T17:04:16.903300Z","shell.execute_reply":"2024-03-12T17:04:16.926225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Creating `Random Forest`, `SVM` and `Decision Tree` Models and comparing them**","metadata":{}},{"cell_type":"code","source":"# Assign X and y\nX = train_data.drop('expert_consensus', axis=1)\ny = train_data['expert_consensus']  ","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.929067Z","iopub.execute_input":"2024-03-12T17:04:16.929408Z","iopub.status.idle":"2024-03-12T17:04:16.940323Z","shell.execute_reply.started":"2024-03-12T17:04:16.929380Z","shell.execute_reply":"2024-03-12T17:04:16.939027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the dataset into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.943917Z","iopub.execute_input":"2024-03-12T17:04:16.944243Z","iopub.status.idle":"2024-03-12T17:04:16.969983Z","shell.execute_reply.started":"2024-03-12T17:04:16.944215Z","shell.execute_reply":"2024-03-12T17:04:16.968985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize models\n# Decision tree\ndecision_tree_model = DecisionTreeClassifier()\n\n# Random forest\nrandom_forest_model = RandomForestClassifier()\n\n# SVM\nsvm_model = SVC()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.971408Z","iopub.execute_input":"2024-03-12T17:04:16.972403Z","iopub.status.idle":"2024-03-12T17:04:16.977613Z","shell.execute_reply.started":"2024-03-12T17:04:16.972343Z","shell.execute_reply":"2024-03-12T17:04:16.976661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train Decision Tree model\ndecision_tree_model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:16.978919Z","iopub.execute_input":"2024-03-12T17:04:16.979510Z","iopub.status.idle":"2024-03-12T17:04:17.723246Z","shell.execute_reply.started":"2024-03-12T17:04:16.979485Z","shell.execute_reply":"2024-03-12T17:04:17.721879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train Random Forest model\nrandom_forest_model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:17.724511Z","iopub.execute_input":"2024-03-12T17:04:17.724821Z","iopub.status.idle":"2024-03-12T17:04:32.494234Z","shell.execute_reply.started":"2024-03-12T17:04:17.724793Z","shell.execute_reply":"2024-03-12T17:04:32.493176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train SVM model\nsvm_model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T17:04:32.495508Z","iopub.execute_input":"2024-03-12T17:04:32.495986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions\ndecision_tree_predictions = decision_tree_model.predict(X_test)\n\nrandom_forest_predictions = random_forest_model.predict(X_test)\n\nsvm_predictions = svm_model.predict(X_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate models\ndecision_tree_accuracy = accuracy_score(y_test, decision_tree_predictions)\n\nrandom_forest_accuracy = accuracy_score(y_test, random_forest_predictions)\n\nsvm_accuracy = accuracy_score(y_test, svm_predictions)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print accuracy of each model\nprint(\"Decision Tree Accuracy:\", decision_tree_accuracy)\nprint(\"Random Forest Accuracy:\", random_forest_accuracy)\nprint(\"SVM Accuracy:\", svm_accuracy)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Determine the best model\nbest_model = max([(decision_tree_accuracy, 'Decision Tree'),\n                  (random_forest_accuracy, 'Random Forest'),\n                  (svm_accuracy, 'SVM')])\n\nprint(\"Best Model:\", best_model[1])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Observations:**\n\n1. There are `no Missing / Null values` in the dataset\n2. There are `no Duplicates` in the dataset\n3. The shape of the dataset is: `(106800, 15)`\n4. The dataset contains both `numerical` and `categorical data`\n5. The dataset contains `10 unique classes of Diseases`\n6. The most common disease is: `Seizures`\n7. The least common disease is: `LPD`\n8. The best model between the 3  models used for prediction is: `Random Forest` with an accuracy score of 99%","metadata":{}}]}