{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Harmful Brain Activity Classification - Naive Bayes Algorithm / Gaussian NB, Multinomial NB, Burnoullis**\n\n## **Project by:** Aarish Asif Khan\n\n## **Date:** 22 February 2024","metadata":{}},{"cell_type":"markdown","source":"# **GaussianNB**","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\n\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.metrics import classification_report","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:41:56.287083Z","iopub.execute_input":"2024-03-10T16:41:56.287703Z","iopub.status.idle":"2024-03-10T16:41:58.983201Z","shell.execute_reply.started":"2024-03-10T16:41:56.287666Z","shell.execute_reply":"2024-03-10T16:41:58.981898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the dataset\ntrain_data = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:41:58.985789Z","iopub.execute_input":"2024-03-10T16:41:58.986419Z","iopub.status.idle":"2024-03-10T16:41:59.278816Z","shell.execute_reply.started":"2024-03-10T16:41:58.986378Z","shell.execute_reply":"2024-03-10T16:41:59.277563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess the data\n# Handle missing values, encode categorical variables, etc.\n\n# Handle Missing Values\nmissing_values = train_data.isnull().sum()\nprint(\"Missing Values:\\n\", missing_values)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:41:59.285576Z","iopub.execute_input":"2024-03-10T16:41:59.286727Z","iopub.status.idle":"2024-03-10T16:41:59.316018Z","shell.execute_reply.started":"2024-03-10T16:41:59.286686Z","shell.execute_reply":"2024-03-10T16:41:59.314916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select features and target\nX = train_data[['eeg_label_offset_seconds', 'spectrogram_label_offset_seconds', 'patient_id']]\ny = train_data['expert_consensus']","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:41:59.317487Z","iopub.execute_input":"2024-03-10T16:41:59.318593Z","iopub.status.idle":"2024-03-10T16:41:59.333909Z","shell.execute_reply.started":"2024-03-10T16:41:59.318553Z","shell.execute_reply":"2024-03-10T16:41:59.332661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train-test split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:41:59.335409Z","iopub.execute_input":"2024-03-10T16:41:59.335884Z","iopub.status.idle":"2024-03-10T16:41:59.364426Z","shell.execute_reply.started":"2024-03-10T16:41:59.335846Z","shell.execute_reply":"2024-03-10T16:41:59.363612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the model\ngnb = GaussianNB()\ngnb.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:41:59.365592Z","iopub.execute_input":"2024-03-10T16:41:59.366586Z","iopub.status.idle":"2024-03-10T16:41:59.657363Z","shell.execute_reply.started":"2024-03-10T16:41:59.366557Z","shell.execute_reply":"2024-03-10T16:41:59.656063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\ny_pred = gnb.predict(X_test)\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:41:59.658769Z","iopub.execute_input":"2024-03-10T16:41:59.659435Z","iopub.status.idle":"2024-03-10T16:42:00.605161Z","shell.execute_reply.started":"2024-03-10T16:41:59.659392Z","shell.execute_reply":"2024-03-10T16:42:00.603966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **MultinomialNB**","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.metrics import classification_report","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:42:00.607089Z","iopub.execute_input":"2024-03-10T16:42:00.607700Z","iopub.status.idle":"2024-03-10T16:42:00.612623Z","shell.execute_reply.started":"2024-03-10T16:42:00.607668Z","shell.execute_reply":"2024-03-10T16:42:00.611613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the dataset\ntrain_data = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:42:00.616755Z","iopub.execute_input":"2024-03-10T16:42:00.617420Z","iopub.status.idle":"2024-03-10T16:42:00.859627Z","shell.execute_reply.started":"2024-03-10T16:42:00.617390Z","shell.execute_reply":"2024-03-10T16:42:00.858007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select features and target\nX = train_data[['eeg_label_offset_seconds', 'spectrogram_label_offset_seconds', 'patient_id']]\ny = train_data['expert_consensus']","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:42:00.861660Z","iopub.execute_input":"2024-03-10T16:42:00.863014Z","iopub.status.idle":"2024-03-10T16:42:00.872948Z","shell.execute_reply.started":"2024-03-10T16:42:00.862955Z","shell.execute_reply":"2024-03-10T16:42:00.871301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train-test split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:42:00.875285Z","iopub.execute_input":"2024-03-10T16:42:00.877287Z","iopub.status.idle":"2024-03-10T16:42:00.908424Z","shell.execute_reply.started":"2024-03-10T16:42:00.877225Z","shell.execute_reply":"2024-03-10T16:42:00.906871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the Multinomial Naive Bayes model\nmnb = MultinomialNB()\nmnb.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:42:00.910353Z","iopub.execute_input":"2024-03-10T16:42:00.911183Z","iopub.status.idle":"2024-03-10T16:42:01.600398Z","shell.execute_reply.started":"2024-03-10T16:42:00.911130Z","shell.execute_reply":"2024-03-10T16:42:01.599185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\ny_pred = mnb.predict(X_test)\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:42:01.602789Z","iopub.execute_input":"2024-03-10T16:42:01.603321Z","iopub.status.idle":"2024-03-10T16:42:02.602445Z","shell.execute_reply.started":"2024-03-10T16:42:01.603284Z","shell.execute_reply":"2024-03-10T16:42:02.601248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Burnoullis**","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.naive_bayes import BernoulliNB\nfrom sklearn.metrics import classification_report","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:42:02.604210Z","iopub.execute_input":"2024-03-10T16:42:02.604915Z","iopub.status.idle":"2024-03-10T16:42:02.610874Z","shell.execute_reply.started":"2024-03-10T16:42:02.604875Z","shell.execute_reply":"2024-03-10T16:42:02.609550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the dataset\ntrain_data = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:42:02.612573Z","iopub.execute_input":"2024-03-10T16:42:02.612892Z","iopub.status.idle":"2024-03-10T16:42:02.810849Z","shell.execute_reply.started":"2024-03-10T16:42:02.612865Z","shell.execute_reply":"2024-03-10T16:42:02.809336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select features and target\nX = train_data[['eeg_label_offset_seconds', 'spectrogram_label_offset_seconds', 'patient_id']]\ny = train_data['expert_consensus']","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:42:02.812140Z","iopub.execute_input":"2024-03-10T16:42:02.812572Z","iopub.status.idle":"2024-03-10T16:42:02.819710Z","shell.execute_reply.started":"2024-03-10T16:42:02.812540Z","shell.execute_reply":"2024-03-10T16:42:02.818688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train-test split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:42:02.820868Z","iopub.execute_input":"2024-03-10T16:42:02.821904Z","iopub.status.idle":"2024-03-10T16:42:02.843002Z","shell.execute_reply.started":"2024-03-10T16:42:02.821872Z","shell.execute_reply":"2024-03-10T16:42:02.841928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the Bernoulli Naive Bayes model\nbnb = BernoulliNB()\nbnb.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:42:02.844287Z","iopub.execute_input":"2024-03-10T16:42:02.844619Z","iopub.status.idle":"2024-03-10T16:42:03.506095Z","shell.execute_reply.started":"2024-03-10T16:42:02.844592Z","shell.execute_reply":"2024-03-10T16:42:03.504910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\ny_pred = bnb.predict(X_test)\nprint(classification_report(y_test, y_pred))","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:42:03.507686Z","iopub.execute_input":"2024-03-10T16:42:03.508056Z","iopub.status.idle":"2024-03-10T16:42:04.522657Z","shell.execute_reply.started":"2024-03-10T16:42:03.508028Z","shell.execute_reply":"2024-03-10T16:42:04.521644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.naive_bayes import GaussianNB, MultinomialNB, BernoulliNB\nfrom sklearn.metrics import accuracy_score\n\n# Load your dataset\ntrain_data = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')  # Replace 'your_dataset.csv' with your actual dataset file path\n\n# Assuming 'X' contains features and 'y' contains labels\nX = train_data.drop(columns=['expert_consensus']) \ny = train_data['expert_consensus']\n\n# Split the dataset into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n# Initialize the three Naive Bayes models\ngnb = GaussianNB()\nmnb = MultinomialNB()\nbnb = BernoulliNB()\n\n# Train the models\ngnb.fit(X_train, y_train)\nmnb.fit(X_train, y_train)\nbnb.fit(X_train, y_train)\n\n# Make predictions\ngnb_pred = gnb.predict(X_test)\nmnb_pred = mnb.predict(X_test)\nbnb_pred = bnb.predict(X_test)\n\n# Calculate accuracy scores\ngnb_accuracy = accuracy_score(y_test, gnb_pred)\nmnb_accuracy = accuracy_score(y_test, mnb_pred)\nbnb_accuracy = accuracy_score(y_test, bnb_pred)\n\n# Plot visualization\nlabels = ['GaussianNB', 'MultinomialNB', 'BernoulliNB']\naccuracies = [gnb_accuracy, mnb_accuracy, bnb_accuracy]\n\nplt.figure(figsize=(10, 6))\nplt.bar(labels, accuracies, color=['blue', 'green', 'red'])\nplt.title('Accuracy Comparison of Naive Bayes Models')\nplt.xlabel('Naive Bayes Models')\nplt.ylabel('Accuracy')\nplt.ylim(0, 1)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:42:04.523774Z","iopub.execute_input":"2024-03-10T16:42:04.524075Z","iopub.status.idle":"2024-03-10T16:42:06.886272Z","shell.execute_reply.started":"2024-03-10T16:42:04.524050Z","shell.execute_reply":"2024-03-10T16:42:06.884606Z"},"trusted":true},"execution_count":null,"outputs":[]}]}