{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:40.016433Z","iopub.execute_input":"2024-01-18T11:12:40.016916Z","iopub.status.idle":"2024-01-18T11:12:40.023379Z","shell.execute_reply.started":"2024-01-18T11:12:40.016882Z","shell.execute_reply":"2024-01-18T11:12:40.021883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\n# Example 2x3 matrix\nmatrix = np.array([[1, 2, 3],\n                   [4, 5, 6]])\n\n# Transpose the matrix and flatten it into a single row\nresult = matrix.T.flatten()\n\nprint(result)\n","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:01:54.002957Z","iopub.execute_input":"2024-01-18T11:01:54.004370Z","iopub.status.idle":"2024-01-18T11:01:54.014216Z","shell.execute_reply.started":"2024-01-18T11:01:54.004323Z","shell.execute_reply":"2024-01-18T11:01:54.012516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ntest = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:42.749909Z","iopub.execute_input":"2024-01-18T11:12:42.750350Z","iopub.status.idle":"2024-01-18T11:12:43.064638Z","shell.execute_reply.started":"2024-01-18T11:12:42.750315Z","shell.execute_reply":"2024-01-18T11:12:43.063388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:44.169847Z","iopub.execute_input":"2024-01-18T11:12:44.170297Z","iopub.status.idle":"2024-01-18T11:12:44.207017Z","shell.execute_reply.started":"2024-01-18T11:12:44.170263Z","shell.execute_reply":"2024-01-18T11:12:44.205802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_dir = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs\"\nspectrogram_dir = \"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms\"","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:44.709733Z","iopub.execute_input":"2024-01-18T11:12:44.710196Z","iopub.status.idle":"2024-01-18T11:12:44.716395Z","shell.execute_reply.started":"2024-01-18T11:12:44.710161Z","shell.execute_reply":"2024-01-18T11:12:44.715260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_files = os.listdir(eeg_dir)\nprint(f\"There are {len(eeg_files)} EEG parquet files\")","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:45.051303Z","iopub.execute_input":"2024-01-18T11:12:45.051727Z","iopub.status.idle":"2024-01-18T11:12:45.238388Z","shell.execute_reply.started":"2024-01-18T11:12:45.051694Z","shell.execute_reply":"2024-01-18T11:12:45.237052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spec_files = os.listdir(spectrogram_dir)\nprint(f\"There are {len(spec_files)} Spectrogram parquet files\")","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:45.240608Z","iopub.execute_input":"2024-01-18T11:12:45.241083Z","iopub.status.idle":"2024-01-18T11:12:45.370921Z","shell.execute_reply.started":"2024-01-18T11:12:45.241041Z","shell.execute_reply":"2024-01-18T11:12:45.369603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:45.428416Z","iopub.execute_input":"2024-01-18T11:12:45.428930Z","iopub.status.idle":"2024-01-18T11:12:45.437456Z","shell.execute_reply.started":"2024-01-18T11:12:45.428888Z","shell.execute_reply":"2024-01-18T11:12:45.435987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = train_df.columns[-6:]\nprint(f\"There are {len(targets)} Targets!\")\nprint(list(targets))","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:45.657745Z","iopub.execute_input":"2024-01-18T11:12:45.658256Z","iopub.status.idle":"2024-01-18T11:12:45.665206Z","shell.execute_reply.started":"2024-01-18T11:12:45.658216Z","shell.execute_reply":"2024-01-18T11:12:45.663853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train_df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:45.872188Z","iopub.execute_input":"2024-01-18T11:12:45.872622Z","iopub.status.idle":"2024-01-18T11:12:45.917166Z","shell.execute_reply.started":"2024-01-18T11:12:45.872589Z","shell.execute_reply":"2024-01-18T11:12:45.915836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = train_df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:46.111680Z","iopub.execute_input":"2024-01-18T11:12:46.112206Z","iopub.status.idle":"2024-01-18T11:12:46.129297Z","shell.execute_reply.started":"2024-01-18T11:12:46.112165Z","shell.execute_reply":"2024-01-18T11:12:46.127880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = train_df.groupby('eeg_id')[['patient_id']].agg('first') # The code adds the patient_id for each eeg_id to the train DataFrame. This links each EEG segment to a specific patient.\ntrain['patient_id'] = tmp","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:46.491077Z","iopub.execute_input":"2024-01-18T11:12:46.491798Z","iopub.status.idle":"2024-01-18T11:12:46.509854Z","shell.execute_reply.started":"2024-01-18T11:12:46.491738Z","shell.execute_reply":"2024-01-18T11:12:46.508640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = train_df.groupby('eeg_id')[targets].agg('sum') # The code sums up the target variable counts (like votes for seizure, LPD, etc.) for each eeg_id.\nfor t in targets:\n    train[t] = tmp[t].values","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:46.791213Z","iopub.execute_input":"2024-01-18T11:12:46.792023Z","iopub.status.idle":"2024-01-18T11:12:46.815253Z","shell.execute_reply.started":"2024-01-18T11:12:46.791971Z","shell.execute_reply":"2024-01-18T11:12:46.813961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_data = train[targets].values # It then normalizes these counts so that they sum up to 1. This step converts the counts into probabilities, which is a common practice in classification tasks.\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[targets] = y_data","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:47.488130Z","iopub.execute_input":"2024-01-18T11:12:47.488567Z","iopub.status.idle":"2024-01-18T11:12:47.498031Z","shell.execute_reply.started":"2024-01-18T11:12:47.488533Z","shell.execute_reply":"2024-01-18T11:12:47.496837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp = train_df.groupby('eeg_id')[['expert_consensus']].agg('first') # For each eeg_id, the code includes the expert_consensus on the EEG segment's classification.\ntrain['target'] = tmp","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:48.077540Z","iopub.execute_input":"2024-01-18T11:12:48.078020Z","iopub.status.idle":"2024-01-18T11:12:48.108161Z","shell.execute_reply.started":"2024-01-18T11:12:48.077981Z","shell.execute_reply":"2024-01-18T11:12:48.106857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.reset_index() # This makes eeg_id a regular column, making the DataFrame easier to work with.\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:48.315327Z","iopub.execute_input":"2024-01-18T11:12:48.316281Z","iopub.status.idle":"2024-01-18T11:12:48.342088Z","shell.execute_reply.started":"2024-01-18T11:12:48.316235Z","shell.execute_reply":"2024-01-18T11:12:48.341051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['eeg_path'] = train['eeg_id'].apply(lambda x: os.path.join(eeg_dir, f'{x}.parquet'))","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:49.243836Z","iopub.execute_input":"2024-01-18T11:12:49.245101Z","iopub.status.idle":"2024-01-18T11:12:49.303666Z","shell.execute_reply.started":"2024-01-18T11:12:49.245049Z","shell.execute_reply":"2024-01-18T11:12:49.302516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['spectrogram_path'] = train['spec_id'].apply(lambda x: os.path.join(spectrogram_dir, f'{x}.parquet'))","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:50.582317Z","iopub.execute_input":"2024-01-18T11:12:50.583669Z","iopub.status.idle":"2024-01-18T11:12:50.642109Z","shell.execute_reply.started":"2024-01-18T11:12:50.583617Z","shell.execute_reply":"2024-01-18T11:12:50.640655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-18T11:12:50.685168Z","iopub.execute_input":"2024-01-18T11:12:50.685651Z","iopub.status.idle":"2024-01-18T11:12:50.708671Z","shell.execute_reply.started":"2024-01-18T11:12:50.685611Z","shell.execute_reply":"2024-01-18T11:12:50.707376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2024-01-18T07:48:06.895992Z","iopub.execute_input":"2024-01-18T07:48:06.896606Z","iopub.status.idle":"2024-01-18T07:48:06.905438Z","shell.execute_reply.started":"2024-01-18T07:48:06.896559Z","shell.execute_reply":"2024-01-18T07:48:06.903835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2024-01-18T07:48:08.713077Z","iopub.execute_input":"2024-01-18T07:48:08.713590Z","iopub.status.idle":"2024-01-18T07:48:08.719538Z","shell.execute_reply.started":"2024-01-18T07:48:08.713548Z","shell.execute_reply":"2024-01-18T07:48:08.718295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_eeg_data(eeg_path):\n    if os.path.isfile(eeg_path):\n        eeg_data = pd.read_parquet(eeg_path)\n        return eeg_data\n    else:\n        print(f\"Invalid file path: {eeg_path}\")\n        return None","metadata":{"execution":{"iopub.status.busy":"2024-01-18T07:48:08.985575Z","iopub.execute_input":"2024-01-18T07:48:08.986085Z","iopub.status.idle":"2024-01-18T07:48:08.994631Z","shell.execute_reply.started":"2024-01-18T07:48:08.986045Z","shell.execute_reply":"2024-01-18T07:48:08.992941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_spectrogram_data(spec_path):\n    if os.path.isfile(spec_path):\n        spec_data = pd.read_parquet(spec_path)\n        return spec_data\n    else:\n        print(f\"Invalid file path: {spec_path}\")\n        return None","metadata":{"execution":{"iopub.status.busy":"2024-01-18T07:48:09.219335Z","iopub.execute_input":"2024-01-18T07:48:09.219813Z","iopub.status.idle":"2024-01-18T07:48:09.226954Z","shell.execute_reply.started":"2024-01-18T07:48:09.219778Z","shell.execute_reply":"2024-01-18T07:48:09.225895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_eeg_data('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1000913311.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-01-18T07:48:09.469400Z","iopub.execute_input":"2024-01-18T07:48:09.470928Z","iopub.status.idle":"2024-01-18T07:48:09.519360Z","shell.execute_reply.started":"2024-01-18T07:48:09.470867Z","shell.execute_reply":"2024-01-18T07:48:09.518093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_spectrogram_data('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000086677.parquet')","metadata":{"execution":{"iopub.status.busy":"2024-01-18T07:42:18.582084Z","iopub.execute_input":"2024-01-18T07:42:18.582886Z","iopub.status.idle":"2024-01-18T07:42:18.689701Z","shell.execute_reply.started":"2024-01-18T07:42:18.582834Z","shell.execute_reply":"2024-01-18T07:42:18.688092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}