{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-16T17:25:15.347803Z","iopub.execute_input":"2024-02-16T17:25:15.348178Z","iopub.status.idle":"2024-02-16T17:25:40.797840Z","shell.execute_reply.started":"2024-02-16T17:25:15.348149Z","shell.execute_reply":"2024-02-16T17:25:40.796657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-16T17:29:25.682651Z","iopub.execute_input":"2024-02-16T17:29:25.683185Z","iopub.status.idle":"2024-02-16T17:29:26.026975Z","shell.execute_reply.started":"2024-02-16T17:29:25.683156Z","shell.execute_reply":"2024-02-16T17:29:26.025847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-16T17:29:26.028770Z","iopub.execute_input":"2024-02-16T17:29:26.029071Z","iopub.status.idle":"2024-02-16T17:29:26.073120Z","shell.execute_reply.started":"2024-02-16T17:29:26.029045Z","shell.execute_reply":"2024-02-16T17:29:26.071947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe(include='all')","metadata":{"execution":{"iopub.status.busy":"2024-02-16T17:29:26.167290Z","iopub.execute_input":"2024-02-16T17:29:26.167725Z","iopub.status.idle":"2024-02-16T17:29:26.297143Z","shell.execute_reply.started":"2024-02-16T17:29:26.167684Z","shell.execute_reply":"2024-02-16T17:29:26.295768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_values = train['expert_consensus'].unique()\nprint(unique_values)","metadata":{"execution":{"iopub.status.busy":"2024-02-16T17:29:26.355070Z","iopub.execute_input":"2024-02-16T17:29:26.355449Z","iopub.status.idle":"2024-02-16T17:29:26.369542Z","shell.execute_reply.started":"2024-02-16T17:29:26.355420Z","shell.execute_reply":"2024-02-16T17:29:26.368352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a mapping\nmapping = {'Seizure': 0, 'GPD': 1, 'LRDA': 2 ,'Other':3,'GRDA':4,'LPD':5}\n\n# Apply the mapping\ntrain['expert_consensus'] = train['expert_consensus'].map(mapping)","metadata":{"execution":{"iopub.status.busy":"2024-02-16T17:29:26.542316Z","iopub.execute_input":"2024-02-16T17:29:26.542875Z","iopub.status.idle":"2024-02-16T17:29:26.561944Z","shell.execute_reply.started":"2024-02-16T17:29:26.542841Z","shell.execute_reply":"2024-02-16T17:29:26.560417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import MinMaxScaler\n\n# Normalization\ncolumns_to_normalize = ['eeg_sub_id','eeg_label_offset_seconds','spectrogram_id','spectrogram_sub_id','spectrogram_label_offset_seconds','label_id','patient_id','expert_consensus',\n                       'seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote']\n\nscaler = MinMaxScaler()\n\n\ntrain[columns_to_normalize] = scaler.fit_transform(train[columns_to_normalize])\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-16T17:30:44.931029Z","iopub.execute_input":"2024-02-16T17:30:44.931448Z","iopub.status.idle":"2024-02-16T17:30:46.559936Z","shell.execute_reply.started":"2024-02-16T17:30:44.931417Z","shell.execute_reply":"2024-02-16T17:30:46.558647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from statsmodels.stats.outliers_influence import variance_inflation_factor\nvariables = train[['eeg_sub_id','eeg_label_offset_seconds','spectrogram_id','spectrogram_sub_id','spectrogram_label_offset_seconds','label_id','patient_id','expert_consensus']]\nvif = pd.DataFrame()\nvif[\"VIF\"] = [variance_inflation_factor(variables.values, i) for i in range(variables.shape[1])]\n# Finally, I like to include names so it is easier to explore the result\nvif[\"Features\"] = variables.columns\nvif","metadata":{"execution":{"iopub.status.busy":"2024-02-16T17:30:46.561989Z","iopub.execute_input":"2024-02-16T17:30:46.562369Z","iopub.status.idle":"2024-02-16T17:30:47.636333Z","shell.execute_reply.started":"2024-02-16T17:30:46.562338Z","shell.execute_reply":"2024-02-16T17:30:47.634602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_no_multicolinearity = train.drop('eeg_sub_id',axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-16T17:30:47.638201Z","iopub.execute_input":"2024-02-16T17:30:47.638764Z","iopub.status.idle":"2024-02-16T17:30:47.662810Z","shell.execute_reply.started":"2024-02-16T17:30:47.638724Z","shell.execute_reply":"2024-02-16T17:30:47.661883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from statsmodels.stats.outliers_influence import variance_inflation_factor\nvariables = data_no_multicolinearity[['eeg_label_offset_seconds','spectrogram_id','spectrogram_sub_id','spectrogram_label_offset_seconds','label_id','patient_id','expert_consensus']]\nvif = pd.DataFrame()\nvif[\"VIF\"] = [variance_inflation_factor(variables.values, i) for i in range(variables.shape[1])]\n# Finally, I like to include names so it is easier to explore the result\nvif[\"Features\"] = variables.columns\nvif","metadata":{"execution":{"iopub.status.busy":"2024-02-16T17:30:47.666097Z","iopub.execute_input":"2024-02-16T17:30:47.666974Z","iopub.status.idle":"2024-02-16T17:30:48.157810Z","shell.execute_reply.started":"2024-02-16T17:30:47.666931Z","shell.execute_reply":"2024-02-16T17:30:48.156710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_no_multicolinearity","metadata":{"execution":{"iopub.status.busy":"2024-02-16T17:30:48.159452Z","iopub.execute_input":"2024-02-16T17:30:48.160124Z","iopub.status.idle":"2024-02-16T17:30:48.220208Z","shell.execute_reply.started":"2024-02-16T17:30:48.160086Z","shell.execute_reply":"2024-02-16T17:30:48.219046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = data_no_multicolinearity.drop(columns=['seizure_vote','lpd_vote','gpd_vote',\n                                           'lrda_vote','grda_vote','other_vote'])\nY = data_no_multicolinearity[['seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote']]\n\nX.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-16T17:30:52.786806Z","iopub.execute_input":"2024-02-16T17:30:52.787185Z","iopub.status.idle":"2024-02-16T17:30:52.814208Z","shell.execute_reply.started":"2024-02-16T17:30:52.787156Z","shell.execute_reply":"2024-02-16T17:30:52.812969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import xgboost as xgb\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import mean_squared_error\n\n# Assuming 'X' contains your features and 'y' contains your target variables\n\n# Split data into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, Y, test_size=0.2, random_state=42)\n\n# Define the XGBoost regressor with default parameters\nxgb_model = xgb.XGBRegressor()\n\n# Fit the model to the training data\nxgb_model.fit(X_train, y_train)\n\n# Predict on the test data\ny_pred = xgb_model.predict(X_test)\n\n# Evaluate the model\nmse = mean_squared_error(y_test, y_pred)\nprint(\"Mean Squared Error:\", mse)\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-02-16T17:30:53.010872Z","iopub.execute_input":"2024-02-16T17:30:53.011296Z","iopub.status.idle":"2024-02-16T17:30:56.364133Z","shell.execute_reply.started":"2024-02-16T17:30:53.011261Z","shell.execute_reply":"2024-02-16T17:30:56.363206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import r2_score\n\n# Assuming y_true are the true target values and y_pred are the predicted values\nr2 = r2_score(y_test, y_pred)\nprint(\"R-squared (R2) score:\", r2)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-16T17:30:56.366005Z","iopub.execute_input":"2024-02-16T17:30:56.366975Z","iopub.status.idle":"2024-02-16T17:30:56.378403Z","shell.execute_reply.started":"2024-02-16T17:30:56.366940Z","shell.execute_reply":"2024-02-16T17:30:56.377332Z"},"trusted":true},"execution_count":null,"outputs":[]}]}