{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 01 Pathing and EDA","metadata":{}},{"cell_type":"code","source":"import os\ndirectory_path = '/kaggle/input'\ncsv_file_paths = []\n\nfor root, dirs, files in os.walk(directory_path):\n    for file in files:\n        if file.endswith(\".csv\"):\n            csv_file_path = os.path.join(root, file)\n            csv_file_paths.append(csv_file_path)\n\nfor csv_path in csv_file_paths:\n    print(csv_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T06:01:09.381745Z","iopub.execute_input":"2024-04-03T06:01:09.382152Z","iopub.status.idle":"2024-04-03T06:01:15.015066Z","shell.execute_reply.started":"2024-04-03T06:01:09.382119Z","shell.execute_reply":"2024-04-03T06:01:15.013942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(\"ignore\")\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\ntrain_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ntest_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nsample_submission = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv')\n                       \ntrain_df","metadata":{"execution":{"iopub.status.busy":"2024-04-03T06:01:15.021465Z","iopub.execute_input":"2024-04-03T06:01:15.021800Z","iopub.status.idle":"2024-04-03T06:01:16.422104Z","shell.execute_reply.started":"2024-04-03T06:01:15.021768Z","shell.execute_reply":"2024-04-03T06:01:16.420774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 01.1: Pie Chart for Data Distribution","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\ntrain_df['expert_consensus'].value_counts().plot(kind='pie', autopct='%1.1f%%')\nplt.title('Distribution of expert_consensus')\nplt.xlabel('Expert Consensus')\nplt.ylabel('Count')\nplt.xticks(rotation=45)\nplt.show()\n\nunique_expert_consensus = train_df['expert_consensus'].unique()\nprint(\"Unique values in 'expert_consensus' column:\", unique_expert_consensus)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T06:01:16.423645Z","iopub.execute_input":"2024-04-03T06:01:16.424612Z","iopub.status.idle":"2024-04-03T06:01:16.711031Z","shell.execute_reply.started":"2024-04-03T06:01:16.424572Z","shell.execute_reply":"2024-04-03T06:01:16.709818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> Check the percentage of each consensus in training data. They are distributed equally in approximation.","metadata":{}},{"cell_type":"markdown","source":"## 01.2: Correlation Heat Map for Consensus","metadata":{}},{"cell_type":"code","source":"train_df_encoded = pd.get_dummies(train_df, columns=['expert_consensus'], drop_first=True)\ncorrelation_matrix = train_df_encoded[['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']].corr()\n\nplt.figure(figsize=(10, 6))\nsns.heatmap(correlation_matrix, annot=True, cmap='coolwarm')\nplt.title('Correlation Heatmap between vote columns and expert_consensus')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T06:01:16.715470Z","iopub.execute_input":"2024-04-03T06:01:16.715841Z","iopub.status.idle":"2024-04-03T06:01:17.289134Z","shell.execute_reply.started":"2024-04-03T06:01:16.715807Z","shell.execute_reply":"2024-04-03T06:01:17.287888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 01.3 Kernel Distribution Plot for Label Second ","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\nsns.kdeplot(train_df['spectrogram_label_offset_seconds'], shade=True)\nplt.title('Distribution of spectrogram_label_offset_seconds (KDE Plot)')\nplt.xlabel('Offset Seconds')\nplt.ylabel('Density')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T06:01:17.290783Z","iopub.execute_input":"2024-04-03T06:01:17.291248Z","iopub.status.idle":"2024-04-03T06:01:18.350127Z","shell.execute_reply.started":"2024-04-03T06:01:17.291208Z","shell.execute_reply":"2024-04-03T06:01:18.348879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vote_columns = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nvote_stats = train_df[vote_columns].describe()\nprint(vote_stats)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-03T06:01:18.351571Z","iopub.execute_input":"2024-04-03T06:01:18.351938Z","iopub.status.idle":"2024-04-03T06:01:18.400433Z","shell.execute_reply.started":"2024-04-03T06:01:18.351893Z","shell.execute_reply":"2024-04-03T06:01:18.399239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 01.4 Box Plot","metadata":{}},{"cell_type":"code","source":"for column in vote_columns:\n    plt.figure(figsize=(8, 5))\n    sns.boxplot(x='expert_consensus', y=column, data=train_df)\n    plt.title(f'Distribution of {column} by expert_consensus')\n    plt.xlabel('expert_consensus')\n    plt.ylabel(column)\n    plt.xticks(rotation=45)\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-03T06:01:18.401877Z","iopub.execute_input":"2024-04-03T06:01:18.402261Z","iopub.status.idle":"2024-04-03T06:01:20.925946Z","shell.execute_reply.started":"2024-04-03T06:01:18.402229Z","shell.execute_reply":"2024-04-03T06:01:20.924609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# [On Progress] 02 PyTorch Modeling","metadata":{}},{"cell_type":"code","source":"import os\ndirectory_path = '/kaggle/input'\nnon_csv_file_paths = []\n\nfor root, dirs, files in os.walk(directory_path):\n    for file in files:\n        if not file.endswith(\".csv\"): \n            file_path = os.path.join(root, file)\n            non_csv_file_paths.append(file_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-03T06:01:20.927349Z","iopub.execute_input":"2024-04-03T06:01:20.927676Z","iopub.status.idle":"2024-04-03T06:01:21.111149Z","shell.execute_reply.started":"2024-04-03T06:01:20.927647Z","shell.execute_reply":"2024-04-03T06:01:21.110152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Length of non_csv_file_paths: {len(non_csv_file_paths)}\")\nprint(f\"Length of train_df: {len(train_df)}\")\nprint(f\"Length of test_df: {len(test_df)}\")\nprint(f\"Length of sample_submission: {len(sample_submission)}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-03T06:01:21.112666Z","iopub.execute_input":"2024-04-03T06:01:21.113025Z","iopub.status.idle":"2024-04-03T06:01:21.119377Z","shell.execute_reply.started":"2024-04-03T06:01:21.112993Z","shell.execute_reply":"2024-04-03T06:01:21.118207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 02.1 Data leakage or Data Overlap","metadata":{}},{"cell_type":"code","source":"overlap_ids = train_df[train_df['patient_id'].isin(test_df['patient_id'])]['patient_id']\ntrain = train_df[~train_df['patient_id'].isin(overlap_ids)]\n\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-04-03T06:01:21.120766Z","iopub.execute_input":"2024-04-03T06:01:21.121164Z","iopub.status.idle":"2024-04-03T06:01:21.165104Z","shell.execute_reply.started":"2024-04-03T06:01:21.121132Z","shell.execute_reply":"2024-04-03T06:01:21.163807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> No data leakage or overlap has been found","metadata":{"execution":{"iopub.status.busy":"2024-02-07T23:40:49.492772Z","iopub.execute_input":"2024-02-07T23:40:49.493614Z","iopub.status.idle":"2024-02-07T23:40:49.511788Z","shell.execute_reply.started":"2024-02-07T23:40:49.493583Z","shell.execute_reply":"2024-02-07T23:40:49.510920Z"}}},{"cell_type":"markdown","source":"## 02.2 Data Split","metadata":{}},{"cell_type":"markdown","source":"> Audio or image data, such as spectrograms, typically take up a large amount of capacity, and it is inefficient to load and process the entire data into memory at once.\n\n> Saving spectrogram data by dividing it into small pieces is a way to efficiently load and process data into memory. This gives you faster performance when processing the entire data by dividing it into small pieces than by processing it all at once.","metadata":{}},{"cell_type":"markdown","source":"## 02.3  Dataset Load","metadata":{}},{"cell_type":"markdown","source":"## 02.4 Train","metadata":{}},{"cell_type":"markdown","source":"from sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom lightgbm import LGBMClassifier\nfrom sklearn.metrics import accuracy_score\nfrom tqdm import tqdm\n\n# Copying the original dataframe to avoid modifying the original data\ntrain_encoded = train.copy()\n\n# Encoding the target variable\nlabel_encoder = LabelEncoder()\ntrain_encoded['expert_consensus'] = label_encoder.fit_transform(train_encoded['expert_consensus'])\n\n# Select features and target variable\nX = train_encoded[['eeg_label_offset_seconds', 'spectrogram_label_offset_seconds', \n           'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote']]\ny = train_encoded['expert_consensus']\n\n# Train-test split\nX_train, X_test, y_train, y_test = train_test_split(X, y, \n                                                    test_size=0.3, \n                                                    random_state=77)\n\n# Initialize and fit model (LightGBM with GPU)\nlgb_Model = LGBMClassifier(device='gpu', random_state=0)\nlgb_Model.fit(X_train, y_train)\n\n# Predictions\ny_train_pred = lgb_Model.predict(X_train)\ny_test_pred = lgb_Model.predict(X_test)\n\n# Evaluate model\ntrain_accuracy = accuracy_score(y_train, y_train_pred)\ntest_accuracy = accuracy_score(y_test, y_test_pred)\n\nprint(f'Train Accuracy (LightGBM with GPU): {train_accuracy:.3f}')\nprint(f'Test Accuracy (LightGBM with GPU): {test_accuracy:.3f}')\n","metadata":{"execution":{"iopub.status.busy":"2024-04-03T06:23:33.869480Z","iopub.execute_input":"2024-04-03T06:23:33.869901Z","iopub.status.idle":"2024-04-03T06:26:05.555272Z","shell.execute_reply.started":"2024-04-03T06:23:33.869868Z","shell.execute_reply":"2024-04-03T06:26:05.553963Z"},"jupyter":{"source_hidden":true}}},{"cell_type":"code","source":"import h2o\nfrom h2o.automl import H2OAutoML\n\n# Initialize H2O\nh2o.init()\n\n# Copying the original dataframe to avoid modifying the original data\ntrain_encoded = train.copy()\n\n# Encoding the target variable\nlabel_encoder = LabelEncoder()\ntrain_encoded['expert_consensus'] = label_encoder.fit_transform(train_encoded['expert_consensus'])\n\n# Convert DataFrame to H2O Frame\ntrain_h2o = h2o.H2OFrame(train_encoded)\n\n# Identify predictors and response\nx = train_h2o.columns[:-1]\ny = train_h2o.columns[-1]\n\n# Train-test split\ntrain_h2o, test_h2o = train_h2o.split_frame(ratios=[0.7], seed=77)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T06:44:49.240643Z","iopub.execute_input":"2024-04-03T06:44:49.241115Z","iopub.status.idle":"2024-04-03T06:44:52.070194Z","shell.execute_reply.started":"2024-04-03T06:44:49.241077Z","shell.execute_reply":"2024-04-03T06:44:52.069095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize and train AutoML\naml = H2OAutoML(max_models=10, seed=1)\naml.train(x=x, y=y, training_frame=train_h2o)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T06:52:39.300286Z","iopub.execute_input":"2024-04-03T06:52:39.301050Z","iopub.status.idle":"2024-04-03T07:09:08.699744Z","shell.execute_reply.started":"2024-04-03T06:52:39.301010Z","shell.execute_reply":"2024-04-03T07:09:08.698601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the best model\nbest_model = aml.leader","metadata":{"execution":{"iopub.status.busy":"2024-04-03T07:09:37.542858Z","iopub.execute_input":"2024-04-03T07:09:37.543342Z","iopub.status.idle":"2024-04-03T07:09:37.557601Z","shell.execute_reply.started":"2024-04-03T07:09:37.543306Z","shell.execute_reply":"2024-04-03T07:09:37.556379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make predictions on test set\ntest_predictions = best_model.predict(test_h2o)\n\n# Evaluate model\nprint(best_model.model_performance(test_data=test_h2o))","metadata":{"execution":{"iopub.status.busy":"2024-04-03T07:09:41.114147Z","iopub.execute_input":"2024-04-03T07:09:41.114618Z","iopub.status.idle":"2024-04-03T07:09:48.559257Z","shell.execute_reply.started":"2024-04-03T07:09:41.114583Z","shell.execute_reply":"2024-04-03T07:09:48.558019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_model","metadata":{"execution":{"iopub.status.busy":"2024-04-03T07:15:32.482421Z","iopub.execute_input":"2024-04-03T07:15:32.482832Z","iopub.status.idle":"2024-04-03T07:15:32.495601Z","shell.execute_reply.started":"2024-04-03T07:15:32.482800Z","shell.execute_reply":"2024-04-03T07:15:32.494420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tqdm\n\n# Iterating over each test data point\nfor i in tqdm.tqdm(range(len(df_test))):\n    # Loading EEG data for a specified eeg_id\n    eeg_id_ = df_test.loc[i, 'eeg_id']\n    tmp = pd.read_parquet(os.path.join(PDIR, 'test_eegs', f'{eeg_id_}.parquet'))\n    \n    # Extracting EEG data from the Cz electrode\n    cz_electrode_data = tmp['Cz']\n    \n    # Adding the extracted data as a row to the testing DataFrame\n    X_test = pd.concat([X_test, cz_electrode_data.reset_index(drop=True).to_frame().transpose()], axis=0)\n\n# Predicting probabilities for each class\npredictions = xgb_Model.predict_proba(X_test)\n\n# Read the sample submission file\nsubmission = pd.read_csv(f'{PDIR}/sample_submission.csv')\n\n# Iterate over each test data point\nfor i in tqdm.tqdm(range(len(df_test))):\n    # Set the 'eeg_id' in the submission DataFrame\n    submission.loc[i, 'eeg_id'] = df_test.loc[i, 'eeg_id']\n    \n    # Set the probability for each class in the submission DataFrame\n    for j, cls_name in enumerate(submission.columns[1:]):\n        submission.loc[i, cls_name] = predictions[i, j]\n","metadata":{"execution":{"iopub.status.busy":"2024-04-03T07:14:27.396273Z","iopub.execute_input":"2024-04-03T07:14:27.397162Z","iopub.status.idle":"2024-04-03T07:14:30.982025Z","shell.execute_reply.started":"2024-04-03T07:14:27.397123Z","shell.execute_reply":"2024-04-03T07:14:30.974345Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Write submission to CSV file\nsubmission.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-03T07:14:32.013839Z","iopub.execute_input":"2024-04-03T07:14:32.014287Z","iopub.status.idle":"2024-04-03T07:14:32.020804Z","shell.execute_reply.started":"2024-04-03T07:14:32.014253Z","shell.execute_reply":"2024-04-03T07:14:32.019723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}