{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<p style=\"font-size: 24px; font-weight: bold;\">Hello there!</p>\n​\n<p style=\"font-size: 16px;\">This notebook introduces a super simple way to create a submission file for the competition of <b>\"HMS - Harmful Brain Activity Classification\"</b>.</p>\n​\n<p style=\"font-size: 16px;\">In this notebook, for classification purposes, we treat only the data from the Cz electrode of EEG signals as features.</p>\n​\n<p style=\"font-size: 16px;\">Essentially, we are using the electrode data itself as features, which implies the need for feature engineering considering frequency characteristics.</p>\n​\n<p style=\"font-size: 16px;\">The purpose of sharing this notebook is to provide a step-by-step guide to creating a submission using as simple code as possible, even if it's a rough implementation.</p>\n​\n<p style=\"font-size: 16px;\">I hope that the release of this notebook will contribute even a little to the excitement of the competition.</p>\n​\n<p style=\"font-size: 16px;\">Let's enjoy Kaggle together!</p>\n​\n<p style=\"font-size: 16px;\">This notebook executes in approximately 30 seconds.</p>\n​\n<h1>Import Modules</h1>","metadata":{}},{"cell_type":"code","source":"import os\nimport tqdm\nimport pandas as pd\nimport numpy as np\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2024-02-28T09:25:31.000970Z","iopub.execute_input":"2024-02-28T09:25:31.001365Z","iopub.status.idle":"2024-02-28T09:25:32.485696Z","shell.execute_reply.started":"2024-02-28T09:25:31.001333Z","shell.execute_reply":"2024-02-28T09:25:32.484511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Global","metadata":{}},{"cell_type":"code","source":"# parent directory\nPDIR = '/kaggle/input/hms-harmful-brain-activity-classification'","metadata":{"execution":{"iopub.status.busy":"2024-02-28T09:25:32.488046Z","iopub.execute_input":"2024-02-28T09:25:32.488533Z","iopub.status.idle":"2024-02-28T09:25:32.493289Z","shell.execute_reply.started":"2024-02-28T09:25:32.488501Z","shell.execute_reply":"2024-02-28T09:25:32.491910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare train data\n\n## Load CSV meta data","metadata":{}},{"cell_type":"code","source":"# Reading the CSV file 'train.csv' located in the directory specified by PDIR\ndf = pd.read_csv(os.path.join(PDIR, 'train.csv'))\n\n# Displaying the first few rows of the DataFrame\ndisplay(df.head())","metadata":{"execution":{"iopub.status.busy":"2024-02-28T09:25:32.495068Z","iopub.execute_input":"2024-02-28T09:25:32.495476Z","iopub.status.idle":"2024-02-28T09:25:32.857728Z","shell.execute_reply.started":"2024-02-28T09:25:32.495447Z","shell.execute_reply":"2024-02-28T09:25:32.856574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load EEG data","metadata":{}},{"cell_type":"code","source":"# Setting the sampling frequency and duration for EEG data collection\nsampling_frequency = 200  # Sampling frequency in Hz\ndata_collection_duration = 50  # Duration of EEG data collection in seconds\ntotal_samples = sampling_frequency * data_collection_duration  # Total number of samples in the duration\n\n# Setting the number of training data points\nnum_train_data_points = 500  \n\n# Creating an empty DataFrame to store training data\ntraining_data_df = pd.DataFrame()\n\n# Iterating over each training data point\nfor i in tqdm.tqdm(range(num_train_data_points)):\n    # Loading EEG data for a specified eeg_id\n    eeg_id = df.loc[i, 'eeg_id']\n    eeg_data = pd.read_parquet(os.path.join(PDIR, 'train_eegs', f'{eeg_id}.parquet'))\n    \n    # Extracting EEG data from the Cz electrode for 50 seconds\n    label_offset_time = df.loc[i, 'eeg_label_offset_seconds']  # Offset time for the EEG label\n    label_offset_index = int(sampling_frequency * label_offset_time)  # Calculating offset index\n    cz_electrode_data = eeg_data['Cz'][label_offset_index:label_offset_index + total_samples]  # Extracting data for Cz electrode\n    \n    # Adding the extracted data as a row to the training DataFrame\n    training_data_df = pd.concat([training_data_df, cz_electrode_data.reset_index(drop=True).to_frame().transpose()], axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T09:25:32.859377Z","iopub.execute_input":"2024-02-28T09:25:32.859695Z","iopub.status.idle":"2024-02-28T09:25:42.411511Z","shell.execute_reply.started":"2024-02-28T09:25:32.859668Z","shell.execute_reply":"2024-02-28T09:25:42.410294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare features (X_train) and target variable (y_train)","metadata":{}},{"cell_type":"code","source":"# Adding diagnosis results\ntraining_data_df['expert_consensus'] = df[:num_train_data_points]['expert_consensus'].values\n\n# Removing rows with missing values\ntraining_data_df = training_data_df.dropna()\ntraining_data_df = training_data_df.reset_index(drop=True)\n\n# Separating data into features and target\ny_train = training_data_df['expert_consensus']\nX_train = training_data_df.drop('expert_consensus', axis=1)\n\n# Displaying the first few rows of the feature  dataset\ndisplay(X_train.head())","metadata":{"execution":{"iopub.status.busy":"2024-02-28T09:25:42.414710Z","iopub.execute_input":"2024-02-28T09:25:42.415110Z","iopub.status.idle":"2024-02-28T09:25:42.486541Z","shell.execute_reply.started":"2024-02-28T09:25:42.415079Z","shell.execute_reply":"2024-02-28T09:25:42.485249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train RandomForestClassifier()","metadata":{}},{"cell_type":"code","source":"# Initializing a RandomForestClassifier with a random state of 0\nforest = RandomForestClassifier(random_state=0)\n\n# Fitting the classifier to the training data\nforest.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T09:25:42.488006Z","iopub.execute_input":"2024-02-28T09:25:42.488860Z","iopub.status.idle":"2024-02-28T09:25:45.340922Z","shell.execute_reply.started":"2024-02-28T09:25:42.488822Z","shell.execute_reply":"2024-02-28T09:25:45.339727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare test Data","metadata":{}},{"cell_type":"markdown","source":"## Load CSV meta data","metadata":{}},{"cell_type":"code","source":"# Reading the CSV file 'test.csv' located in the directory specified by PDIR\ndf_test = pd.read_csv(os.path.join(PDIR, 'test.csv'))\n\n# Displaying the first few rows of the DataFrame\ndisplay(df_test.head())","metadata":{"execution":{"iopub.status.busy":"2024-02-28T09:25:45.342617Z","iopub.execute_input":"2024-02-28T09:25:45.343615Z","iopub.status.idle":"2024-02-28T09:25:45.361018Z","shell.execute_reply.started":"2024-02-28T09:25:45.343572Z","shell.execute_reply":"2024-02-28T09:25:45.359519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare features (X_test) ","metadata":{}},{"cell_type":"code","source":"# Creating an empty DataFrame to store testing data\nX_test = pd.DataFrame()\n\n# Iterating over each test data point\nfor i in tqdm.tqdm(range(len(df_test))):\n    # Loading EEG data for a specified eeg_id\n    eeg_id_ = df_test.loc[i, 'eeg_id']\n    tmp = pd.read_parquet(os.path.join(PDIR, 'test_eegs', f'{eeg_id_}.parquet'))\n    \n    # Extracting EEG data from the Cz electrode\n    cz_electrode_data = tmp['Cz']\n    \n    # Adding the extracted data as a row to the testing DataFrame\n    X_test = pd.concat([X_test, cz_electrode_data.reset_index(drop=True).to_frame().transpose()], axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T09:25:45.362585Z","iopub.execute_input":"2024-02-28T09:25:45.363010Z","iopub.status.idle":"2024-02-28T09:25:45.409358Z","shell.execute_reply.started":"2024-02-28T09:25:45.362975Z","shell.execute_reply":"2024-02-28T09:25:45.407114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict and submit","metadata":{}},{"cell_type":"code","source":"# Calculate predictions using the trained RandomForestClassifier model\npredictions = forest.predict_proba(X_test)\n\n# Read the sample submission file\nsubmission = pd.read_csv(f'{PDIR}/sample_submission.csv')\n\n# Iterate over each test data point\nfor i in tqdm.tqdm(range(len(df_test))):\n    # Set the 'eeg_id' in the submission DataFrame\n    submission.loc[i, 'eeg_id'] = df_test.loc[i, 'eeg_id']\n    \n    # Set the probability for each class in the submission DataFrame\n    for j, cls_name in enumerate(forest.classes_):\n        submission.loc[i, f'{cls_name.lower()}_vote'] = predictions[i, j]","metadata":{"execution":{"iopub.status.busy":"2024-02-28T09:25:45.411032Z","iopub.execute_input":"2024-02-28T09:25:45.411476Z","iopub.status.idle":"2024-02-28T09:25:45.618047Z","shell.execute_reply.started":"2024-02-28T09:25:45.411437Z","shell.execute_reply":"2024-02-28T09:25:45.616847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display the submission DataFrame\ndisplay(submission)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T09:25:45.619820Z","iopub.execute_input":"2024-02-28T09:25:45.620286Z","iopub.status.idle":"2024-02-28T09:25:45.635790Z","shell.execute_reply.started":"2024-02-28T09:25:45.620240Z","shell.execute_reply":"2024-02-28T09:25:45.634308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving the submission DataFrame to a CSV file without including the index\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-28T09:25:45.637535Z","iopub.execute_input":"2024-02-28T09:25:45.638280Z","iopub.status.idle":"2024-02-28T09:25:45.649406Z","shell.execute_reply.started":"2024-02-28T09:25:45.638235Z","shell.execute_reply":"2024-02-28T09:25:45.647955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Congratulations!\n\nYou're now ready to submit your work on Kaggle!\n\nEnjoy your experience on Kaggle!","metadata":{}}]}