{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Harmful Brain Activity Classification / Lasso Regression**\n\n## **Written by:** Aarish Asif Khan\n\n## **Date:** 25 February 2024","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt \nimport seaborn as sns \n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler, LabelEncoder, OrdinalEncoder\n\nfrom sklearn.linear_model import Lasso\nfrom sklearn.metrics import accuracy_score, classification_report","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:33:06.755547Z","iopub.execute_input":"2024-03-10T16:33:06.755930Z","iopub.status.idle":"2024-03-10T16:33:06.763624Z","shell.execute_reply.started":"2024-03-10T16:33:06.755900Z","shell.execute_reply":"2024-03-10T16:33:06.761855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the dataset\ntrain_data = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Print the 5 rows of the dataset\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:33:06.766863Z","iopub.execute_input":"2024-03-10T16:33:06.767337Z","iopub.status.idle":"2024-03-10T16:33:07.035188Z","shell.execute_reply.started":"2024-03-10T16:33:06.767301Z","shell.execute_reply":"2024-03-10T16:33:07.033966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize LabelEncoder\nlabel_encoder = LabelEncoder()\n\n# Apply label encoding to the expert_consensus column\ntrain_data['expert_consensus_encoded'] = label_encoder.fit_transform(train_data['expert_consensus'])\n\n# Check the encoded values\nencoded_values = train_data[['expert_consensus', 'expert_consensus_encoded']].drop_duplicates()\nprint(encoded_values)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:33:07.037347Z","iopub.execute_input":"2024-03-10T16:33:07.037837Z","iopub.status.idle":"2024-03-10T16:33:07.094547Z","shell.execute_reply.started":"2024-03-10T16:33:07.037804Z","shell.execute_reply":"2024-03-10T16:33:07.093437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign values to X and y\nX = train_data[['eeg_id', 'eeg_sub_id', 'eeg_label_offset_seconds', 'spectrogram_id', 'spectrogram_sub_id', 'spectrogram_label_offset_seconds', 'label_id', 'patient_id']]\ny = train_data['expert_consensus_encoded']","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:33:07.095860Z","iopub.execute_input":"2024-03-10T16:33:07.096350Z","iopub.status.idle":"2024-03-10T16:33:07.105319Z","shell.execute_reply.started":"2024-03-10T16:33:07.096319Z","shell.execute_reply":"2024-03-10T16:33:07.104136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train the Lasso regression model\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:33:07.110025Z","iopub.execute_input":"2024-03-10T16:33:07.110760Z","iopub.status.idle":"2024-03-10T16:33:07.128891Z","shell.execute_reply.started":"2024-03-10T16:33:07.110727Z","shell.execute_reply":"2024-03-10T16:33:07.127370Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Apply standard scaler\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:33:07.130651Z","iopub.execute_input":"2024-03-10T16:33:07.131030Z","iopub.status.idle":"2024-03-10T16:33:07.151864Z","shell.execute_reply.started":"2024-03-10T16:33:07.131000Z","shell.execute_reply":"2024-03-10T16:33:07.150616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initalize and train the model\nlasso = Lasso(alpha=0.1)\nlasso.fit(X_train_scaled, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:33:07.155869Z","iopub.execute_input":"2024-03-10T16:33:07.156560Z","iopub.status.idle":"2024-03-10T16:33:07.196445Z","shell.execute_reply.started":"2024-03-10T16:33:07.156529Z","shell.execute_reply":"2024-03-10T16:33:07.194943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model\ny_pred = lasso.predict(X_test_scaled)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:33:07.199139Z","iopub.execute_input":"2024-03-10T16:33:07.200235Z","iopub.status.idle":"2024-03-10T16:33:07.221700Z","shell.execute_reply.started":"2024-03-10T16:33:07.200180Z","shell.execute_reply":"2024-03-10T16:33:07.219714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert predictions to binary (0 or 1)\ny_pred_binary = np.where(y_pred > 0.5, 1, 0)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:33:07.225071Z","iopub.execute_input":"2024-03-10T16:33:07.227294Z","iopub.status.idle":"2024-03-10T16:33:07.237310Z","shell.execute_reply.started":"2024-03-10T16:33:07.227228Z","shell.execute_reply":"2024-03-10T16:33:07.235699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate accuracy\naccuracy = accuracy_score(y_test, y_pred_binary)\n\n# Generate classification report\nreport = classification_report(y_test, y_pred_binary, zero_division=1)  # You can adjust zero_division to your preference\n\n# Print results\nprint(\"Accuracy:\", accuracy)\nprint(\"Classification Report:\")\nprint(report)","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:33:07.241651Z","iopub.execute_input":"2024-03-10T16:33:07.244154Z","iopub.status.idle":"2024-03-10T16:33:07.349096Z","shell.execute_reply.started":"2024-03-10T16:33:07.244096Z","shell.execute_reply":"2024-03-10T16:33:07.348255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Count the frequency of each encoded label\nlabel_counts = train_data['expert_consensus_encoded'].value_counts().sort_index()\n\n# Define colors for the bars\ncolors = ['cyan', 'purple', 'red', 'orange', 'yellow', 'blue']\n\n# Plot the bar plot with specified colors and thinner bars\nplt.figure(figsize=(10, 6))\nsns.barplot(x=label_counts.index, y=label_counts.values, linewidth=1, palette=colors)\nplt.title('Distribution of Encoded Labels')\nplt.xlabel('Encoded Label')\nplt.ylabel('Frequency')\n\n# Add gridlines\nplt.grid(True, axis='y')\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-10T16:33:07.351695Z","iopub.execute_input":"2024-03-10T16:33:07.352597Z","iopub.status.idle":"2024-03-10T16:33:07.680959Z","shell.execute_reply.started":"2024-03-10T16:33:07.352566Z","shell.execute_reply":"2024-03-10T16:33:07.680124Z"},"trusted":true},"execution_count":null,"outputs":[]}]}