{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Harmful Brain Activity Classification - Polynomial Regression model**\n\n## **Written by:** [Aarish Asif Khan](https://www.kaggle.com/aarishasifkhan)\n\n## **Date:** 24 February 2024\n\n## **Dataset:** [HMS - Harmful Brain Activity Dataset](https://www.kaggle.com/competitions/hms-harmful-brain-activity-classification/overview)","metadata":{}},{"cell_type":"code","source":"# Import libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns \n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import PolynomialFeatures, LabelEncoder\n\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.metrics import mean_absolute_error, mean_squared_error","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:56:34.749333Z","iopub.execute_input":"2024-03-18T12:56:34.751702Z","iopub.status.idle":"2024-03-18T12:56:38.295280Z","shell.execute_reply.started":"2024-03-18T12:56:34.751664Z","shell.execute_reply":"2024-03-18T12:56:38.293843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the dataset\ntrain_data = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Print the 5 rows of the datasets\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:56:38.297787Z","iopub.execute_input":"2024-03-18T12:56:38.298500Z","iopub.status.idle":"2024-03-18T12:56:38.635166Z","shell.execute_reply.started":"2024-03-18T12:56:38.298455Z","shell.execute_reply":"2024-03-18T12:56:38.633926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select features and target\nX = train_data[['eeg_label_offset_seconds', 'spectrogram_label_offset_seconds']].values\ny = train_data['expert_consensus'].values","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:56:38.636710Z","iopub.execute_input":"2024-03-18T12:56:38.637140Z","iopub.status.idle":"2024-03-18T12:56:38.652614Z","shell.execute_reply.started":"2024-03-18T12:56:38.637097Z","shell.execute_reply":"2024-03-18T12:56:38.650728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:56:38.656126Z","iopub.execute_input":"2024-03-18T12:56:38.657520Z","iopub.status.idle":"2024-03-18T12:56:38.681187Z","shell.execute_reply.started":"2024-03-18T12:56:38.657469Z","shell.execute_reply":"2024-03-18T12:56:38.679653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Polynomial regression\ndegree = 4  # Degree of the polynomial\n\npoly_features = PolynomialFeatures(degree=degree)\nX_train_poly = poly_features.fit_transform(X_train)\nX_test_poly = poly_features.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:56:38.683313Z","iopub.execute_input":"2024-03-18T12:56:38.683725Z","iopub.status.idle":"2024-03-18T12:56:38.712889Z","shell.execute_reply.started":"2024-03-18T12:56:38.683690Z","shell.execute_reply":"2024-03-18T12:56:38.711640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encode categorical labels into numerical values\nlabel_encoder = LabelEncoder()\ny_train_encoded = label_encoder.fit_transform(y_train)\ny_test_encoded = label_encoder.transform(y_test)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:56:38.716269Z","iopub.execute_input":"2024-03-18T12:56:38.716725Z","iopub.status.idle":"2024-03-18T12:56:38.757685Z","shell.execute_reply.started":"2024-03-18T12:56:38.716691Z","shell.execute_reply":"2024-03-18T12:56:38.755592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize and train the model\nmodel = LinearRegression()\nmodel.fit(X_train_poly, y_train_encoded)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:56:38.759539Z","iopub.execute_input":"2024-03-18T12:56:38.760288Z","iopub.status.idle":"2024-03-18T12:56:38.880572Z","shell.execute_reply.started":"2024-03-18T12:56:38.760244Z","shell.execute_reply":"2024-03-18T12:56:38.879088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Predictions on training and testing data\ntrain_predictions = model.predict(X_train_poly)\ntest_predictions = model.predict(X_test_poly)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:56:38.882751Z","iopub.execute_input":"2024-03-18T12:56:38.885024Z","iopub.status.idle":"2024-03-18T12:56:38.921766Z","shell.execute_reply.started":"2024-03-18T12:56:38.884964Z","shell.execute_reply":"2024-03-18T12:56:38.919958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate mean absolute error (MAE) and mean squared error (MSE)\ntrain_mae = mean_absolute_error(y_train_encoded, train_predictions)\ntest_mae = mean_absolute_error(y_test_encoded, test_predictions)\n\ntrain_mse = mean_squared_error(y_train_encoded, train_predictions)\ntest_mse = mean_squared_error(y_test_encoded, test_predictions)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:56:38.924026Z","iopub.execute_input":"2024-03-18T12:56:38.924745Z","iopub.status.idle":"2024-03-18T12:56:38.947835Z","shell.execute_reply.started":"2024-03-18T12:56:38.924684Z","shell.execute_reply":"2024-03-18T12:56:38.946170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_mse = mean_squared_error(y_train_encoded, train_predictions)\ntest_mse = mean_squared_error(y_test_encoded, test_predictions)","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:56:38.953189Z","iopub.execute_input":"2024-03-18T12:56:38.959018Z","iopub.status.idle":"2024-03-18T12:56:38.971393Z","shell.execute_reply.started":"2024-03-18T12:56:38.958951Z","shell.execute_reply":"2024-03-18T12:56:38.969342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Print metrics\nprint(\"Metrics:\")\nprint(f'Train MAE: {train_mae}')\nprint(f'Test MAE: {test_mae}')\nprint(f'Train MSE: {train_mse}')\nprint(f'Test MSE: {test_mse}')","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:56:38.974466Z","iopub.execute_input":"2024-03-18T12:56:38.975614Z","iopub.status.idle":"2024-03-18T12:56:38.992934Z","shell.execute_reply.started":"2024-03-18T12:56:38.975556Z","shell.execute_reply":"2024-03-18T12:56:38.989973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualize the performance\nplt.scatter(y_train, train_predictions, color='blue', label='Train')\nplt.scatter(y_test, test_predictions, color='red', label='Test')\nplt.xlabel('Actual')\nplt.ylabel('Predicted')\nplt.title('Polynomial Regression Performance')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:56:38.998341Z","iopub.execute_input":"2024-03-18T12:56:39.000047Z","iopub.status.idle":"2024-03-18T12:56:41.526132Z","shell.execute_reply.started":"2024-03-18T12:56:38.999982Z","shell.execute_reply":"2024-03-18T12:56:41.524757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Encode categorical labels into numerical values for y_train and y_test\nlabel_encoder = LabelEncoder()\ny_train_encoded = label_encoder.fit_transform(y_train)\ny_test_encoded = label_encoder.transform(y_test)\n\ndegrees = [1, 2, 3, 4, 5]\nbest_degree = None\nbest_mae = float('inf')\n\nfor degree in degrees:\n    # Create polynomial features\n    poly_features = PolynomialFeatures(degree=degree)\n    X_train_poly = poly_features.fit_transform(X_train)\n    X_test_poly = poly_features.transform(X_test)\n    \n    # Train the model\n    model = LinearRegression()\n    model.fit(X_train_poly, y_train_encoded)\n    \n    # Predictions\n    test_predictions = model.predict(X_test_poly)\n    \n    # Calculate MAE\n    mae = mean_absolute_error(y_test_encoded, test_predictions)\n    \n    # Update best degree if lower MAE found\n    if mae < best_mae:\n        best_mae = mae\n        best_degree = degree\n\nprint(\"Best degree:\", best_degree)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-18T12:56:41.527347Z","iopub.execute_input":"2024-03-18T12:56:41.527698Z","iopub.status.idle":"2024-03-18T12:56:42.027811Z","shell.execute_reply.started":"2024-03-18T12:56:41.527669Z","shell.execute_reply":"2024-03-18T12:56:42.026369Z"},"trusted":true},"execution_count":null,"outputs":[]}]}