{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Harmful Brain Activity Classification - Linear Regression Model**\n\n## **Project by:** [Aarish Asif Khan](https://www.kaggle.com/aarishasifkhan)\n\n## **Date:** 11th February 2024\n\n## **Dataset:** [HMS - Harmful Brain Activity Dataset](https://www.kaggle.com/competitions/hms-harmful-brain-activity-classification)","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport pandas as pd \nimport numpy as np \nimport matplotlib.pyplot as plt \nimport seaborn as sns \n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_squared_error, r2_score\nfrom sklearn.preprocessing import StandardScaler","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:45.730923Z","iopub.execute_input":"2024-03-18T08:08:45.731572Z","iopub.status.idle":"2024-03-18T08:08:48.873587Z","shell.execute_reply.started":"2024-03-18T08:08:45.731537Z","shell.execute_reply":"2024-03-18T08:08:48.872347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the HMC dataset\ntrain_data = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:48.875814Z","iopub.execute_input":"2024-03-18T08:08:48.876461Z","iopub.status.idle":"2024-03-18T08:08:49.278567Z","shell.execute_reply.started":"2024-03-18T08:08:48.876415Z","shell.execute_reply":"2024-03-18T08:08:49.277205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assign values to X and y\nX = train_data[['eeg_label_offset_seconds']]\n\ny = pd.get_dummies(train_data['expert_consensus'])","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.285868Z","iopub.execute_input":"2024-03-18T08:08:49.286311Z","iopub.status.idle":"2024-03-18T08:08:49.319710Z","shell.execute_reply.started":"2024-03-18T08:08:49.286270Z","shell.execute_reply":"2024-03-18T08:08:49.318750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the data into training and testing with train-test-split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.320828Z","iopub.execute_input":"2024-03-18T08:08:49.321814Z","iopub.status.idle":"2024-03-18T08:08:49.342830Z","shell.execute_reply.started":"2024-03-18T08:08:49.321782Z","shell.execute_reply":"2024-03-18T08:08:49.341504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if there are any missing values\nprint(\"Number of missing values in X_train:\", X_train.isnull().sum())\nprint(\"Number of missing values in y_train:\", y_train.isnull().sum())\n","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.344619Z","iopub.execute_input":"2024-03-18T08:08:49.345092Z","iopub.status.idle":"2024-03-18T08:08:49.354770Z","shell.execute_reply.started":"2024-03-18T08:08:49.345042Z","shell.execute_reply":"2024-03-18T08:08:49.353896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build linear regression model\nmodel = LinearRegression()\n\n# train the model\nmodel.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.355960Z","iopub.execute_input":"2024-03-18T08:08:49.356458Z","iopub.status.idle":"2024-03-18T08:08:49.423954Z","shell.execute_reply.started":"2024-03-18T08:08:49.356430Z","shell.execute_reply":"2024-03-18T08:08:49.422696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model evaluation: predict the model\ny_predict = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.425384Z","iopub.execute_input":"2024-03-18T08:08:49.425749Z","iopub.status.idle":"2024-03-18T08:08:49.434435Z","shell.execute_reply.started":"2024-03-18T08:08:49.425714Z","shell.execute_reply":"2024-03-18T08:08:49.433109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mse = mean_squared_error(y_test, y_predict)\nr2 = r2_score(y_test, y_predict)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.436180Z","iopub.execute_input":"2024-03-18T08:08:49.436654Z","iopub.status.idle":"2024-03-18T08:08:49.462749Z","shell.execute_reply.started":"2024-03-18T08:08:49.436607Z","shell.execute_reply":"2024-03-18T08:08:49.461050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Mean Squared Error (MSE):\", mse)\nprint(\"R-squared (R2) Score:\", r2)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.470640Z","iopub.execute_input":"2024-03-18T08:08:49.471610Z","iopub.status.idle":"2024-03-18T08:08:49.477272Z","shell.execute_reply.started":"2024-03-18T08:08:49.471563Z","shell.execute_reply":"2024-03-18T08:08:49.476190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data pre-processing\nscaler = StandardScaler()\n\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.478708Z","iopub.execute_input":"2024-03-18T08:08:49.479807Z","iopub.status.idle":"2024-03-18T08:08:49.503488Z","shell.execute_reply.started":"2024-03-18T08:08:49.479761Z","shell.execute_reply":"2024-03-18T08:08:49.502110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Retrain the model with scaled features\nmodel_scaled = LinearRegression()\nmodel_scaled.fit(X_train_scaled, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.505391Z","iopub.execute_input":"2024-03-18T08:08:49.506279Z","iopub.status.idle":"2024-03-18T08:08:49.541402Z","shell.execute_reply.started":"2024-03-18T08:08:49.506237Z","shell.execute_reply":"2024-03-18T08:08:49.540302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluate the model with scaled features\ny_pred_scaled = model_scaled.predict(X_test_scaled)\nmse_scaled = mean_squared_error(y_test, y_pred_scaled)\nr2_scaled = r2_score(y_test, y_pred_scaled)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.543052Z","iopub.execute_input":"2024-03-18T08:08:49.543799Z","iopub.status.idle":"2024-03-18T08:08:49.561721Z","shell.execute_reply.started":"2024-03-18T08:08:49.543759Z","shell.execute_reply":"2024-03-18T08:08:49.560376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Mean Squared Error (MSE) with feature scaling:\", mse_scaled)\nprint(\"R-squared (R2) Score with feature scaling:\", r2_scaled)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.563806Z","iopub.execute_input":"2024-03-18T08:08:49.564635Z","iopub.status.idle":"2024-03-18T08:08:49.571616Z","shell.execute_reply.started":"2024-03-18T08:08:49.564587Z","shell.execute_reply":"2024-03-18T08:08:49.570337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"coefficients = model.coef_\n\n# Print the coefficients\nprint(\"Coefficients:\", coefficients)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.573930Z","iopub.execute_input":"2024-03-18T08:08:49.574831Z","iopub.status.idle":"2024-03-18T08:08:49.588211Z","shell.execute_reply.started":"2024-03-18T08:08:49.574784Z","shell.execute_reply":"2024-03-18T08:08:49.587090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the residuals\nresiduals = y_test - y_predict\n\n# Print the residuals\nprint(\"Residuals:\", residuals)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.590333Z","iopub.execute_input":"2024-03-18T08:08:49.591377Z","iopub.status.idle":"2024-03-18T08:08:49.607114Z","shell.execute_reply.started":"2024-03-18T08:08:49.591330Z","shell.execute_reply":"2024-03-18T08:08:49.605900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Access the intercept\nintercept = model.intercept_\n\n# Print the intercept\nprint(\"Intercept:\", intercept)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.609735Z","iopub.execute_input":"2024-03-18T08:08:49.610598Z","iopub.status.idle":"2024-03-18T08:08:49.618531Z","shell.execute_reply.started":"2024-03-18T08:08:49.610565Z","shell.execute_reply":"2024-03-18T08:08:49.617170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.histplot(train_data['eeg_label_offset_seconds'], bins=20, kde=True, color='blue')\nplt.xlabel('EEG Label Offset (Seconds)')\nplt.ylabel('Frequency')\nplt.title('Histogram of EEG Label Offset (Seconds)')\nplt.show()\n\n# Create scatter plots for the relationship between the feature variable and each target variable\nplt.figure(figsize=(15, 10))\nfor i, target_col in enumerate(y.columns):\n    plt.subplot(2, 3, i + 1)\n    sns.scatterplot(x=X['eeg_label_offset_seconds'], y=y[target_col], color='green')\n    plt.xlabel('EEG Label Offset (Seconds)')\n    plt.ylabel(target_col)\n    plt.title(f'Scatter Plot: EEG Label Offset vs {target_col}')\n\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-03-18T08:08:49.620805Z","iopub.execute_input":"2024-03-18T08:08:49.621585Z","iopub.status.idle":"2024-03-18T08:08:53.960538Z","shell.execute_reply.started":"2024-03-18T08:08:49.621544Z","shell.execute_reply":"2024-03-18T08:08:53.959437Z"},"trusted":true},"execution_count":null,"outputs":[]}]}