{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd, numpy as np\nfrom tqdm import tqdm\nimport pickle\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-13T10:12:36.959199Z","iopub.execute_input":"2024-01-13T10:12:36.959903Z","iopub.status.idle":"2024-01-13T10:12:37.540074Z","shell.execute_reply.started":"2024-01-13T10:12:36.959841Z","shell.execute_reply":"2024-01-13T10:12:37.538813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-13T10:12:37.542418Z","iopub.execute_input":"2024-01-13T10:12:37.542960Z","iopub.status.idle":"2024-01-13T10:12:37.899238Z","shell.execute_reply.started":"2024-01-13T10:12:37.542921Z","shell.execute_reply":"2024-01-13T10:12:37.898329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the dataset description, we can know:\n* Each patient's data is split into several subsamples.\n* Each subsample covers 50 seconds.\n\nWe can use the **'eeg_label_offset_seconds'** to split off each subsample.","metadata":{}},{"cell_type":"code","source":"# here is an exmaple, patient_id=30631 has 270 eeg_ids, we pick one 2211351621.\ndf_example = df[(df.patient_id==30631)&(df.eeg_id==2211351621)].reset_index(drop=True)\ndisplay(df_example)","metadata":{"execution":{"iopub.status.busy":"2024-01-13T10:12:37.900790Z","iopub.execute_input":"2024-01-13T10:12:37.901163Z","iopub.status.idle":"2024-01-13T10:12:37.927287Z","shell.execute_reply.started":"2024-01-13T10:12:37.901131Z","shell.execute_reply":"2024-01-13T10:12:37.926111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we can see **eeg_label_offset_seconds** includes [0, 6, 10, 34, 68, 78, 82].\nLoad its EEG data.","metadata":{}},{"cell_type":"code","source":"eeg_example = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/2211351621.parquet')\nprint(f'eeg_id=2211351621 has {eeg_example.shape[0]} samples.')\nprint()\ndisplay(eeg_example.head())","metadata":{"execution":{"iopub.status.busy":"2024-01-13T10:12:37.929038Z","iopub.execute_input":"2024-01-13T10:12:37.929564Z","iopub.status.idle":"2024-01-13T10:12:38.056627Z","shell.execute_reply.started":"2024-01-13T10:12:37.929513Z","shell.execute_reply":"2024-01-13T10:12:38.055427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The range of eeg_sub_id=0\nstart_0, end_0 = 0*200, (0+50)*200\nprint(f\"eeg_sub_id=0 range from {start_0} to {end_0}\")\n\n# The range of eeg_sub_id=1\nstart_1, end_1 = 6*200, (6+50)*200\nprint(f\"eeg_sub_id=1 range from {start_1} to {end_1}\")\n\n# The range of eeg_sub_id=2\nstart_2, end_2 = 10*200, (10+50)*200\nprint(f\"eeg_sub_id=2 range from {start_2} to {end_2}\")\n\n# The range of eeg_sub_id=3\nstart_3, end_3 = 34*200, (34+50)*200\nprint(f\"eeg_sub_id=3 range from {start_3} to {end_3}\")\n\n# The range of eeg_sub_id=4\nstart_4, end_4 = 68*200, (68+50)*200\nprint(f\"eeg_sub_id=4 range from {start_4} to {end_4}\")\n\n# The range of eeg_sub_id=5\nstart_5, end_5 = 78*200, (78+50)*200\nprint(f\"eeg_sub_id=5 range from {start_5} to {end_5}\")\n\n# The range of eeg_sub_id=6\nstart_6, end_6 = 82*200, (82+50)*200\nprint(f\"eeg_sub_id=6 range from {start_6} to {end_6}\")\n\n","metadata":{"execution":{"iopub.status.busy":"2024-01-13T10:12:38.059841Z","iopub.execute_input":"2024-01-13T10:12:38.060848Z","iopub.status.idle":"2024-01-13T10:12:38.072746Z","shell.execute_reply.started":"2024-01-13T10:12:38.060807Z","shell.execute_reply":"2024-01-13T10:12:38.070334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The EEG data of eeg_id=2211351621 has 26400 samples and the last subsample ends at 26400.","metadata":{}},{"cell_type":"code","source":"# here is an exmaple, patient_id=55045 has 1 eeg_id 3449810742.\ndf_example = df[(df.patient_id==55045)].reset_index(drop=True)\ndisplay(df_example)","metadata":{"execution":{"iopub.status.busy":"2024-01-13T10:12:38.074426Z","iopub.execute_input":"2024-01-13T10:12:38.075632Z","iopub.status.idle":"2024-01-13T10:12:38.098018Z","shell.execute_reply.started":"2024-01-13T10:12:38.075586Z","shell.execute_reply":"2024-01-13T10:12:38.096777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we can see **eeg_label_offset_seconds** includes [0, 4, 10, 12, 14, 24].\nLoad its EEG data.","metadata":{}},{"cell_type":"code","source":"eeg_example = pd.read_parquet('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/3449810742.parquet')\nprint(f'eeg_id=3449810742 has {eeg_example.shape[0]} samples.')\nprint()\ndisplay(eeg_example.head())","metadata":{"execution":{"iopub.status.busy":"2024-01-13T10:12:38.100057Z","iopub.execute_input":"2024-01-13T10:12:38.100493Z","iopub.status.idle":"2024-01-13T10:12:38.139965Z","shell.execute_reply.started":"2024-01-13T10:12:38.100455Z","shell.execute_reply":"2024-01-13T10:12:38.139124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The range of eeg_sub_id=0\nstart_0, end_0 = 0*200, (0+50)*200\nprint(f\"eeg_sub_id=0 range from {start_0} to {end_0}\")\n\n# The range of eeg_sub_id=1\nstart_1, end_1 = 4*200, (4+50)*200\nprint(f\"eeg_sub_id=1 range from {start_1} to {end_1}\")\n\n# The range of eeg_sub_id=2\nstart_2, end_2 = 10*200, (10+50)*200\nprint(f\"eeg_sub_id=2 range from {start_2} to {end_2}\")\n\n# The range of eeg_sub_id=3\nstart_3, end_3 = 12*200, (12+50)*200\nprint(f\"eeg_sub_id=3 range from {start_3} to {end_3}\")\n\n# The range of eeg_sub_id=4\nstart_4, end_4 = 14*200, (14+50)*200\nprint(f\"eeg_sub_id=4 range from {start_4} to {end_4}\")\n\n# The range of eeg_sub_id=5\nstart_5, end_5 = 24*200, (24+50)*200\nprint(f\"eeg_sub_id=5 range from {start_5} to {end_5}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-13T10:12:38.141306Z","iopub.execute_input":"2024-01-13T10:12:38.141853Z","iopub.status.idle":"2024-01-13T10:12:38.151196Z","shell.execute_reply.started":"2024-01-13T10:12:38.141818Z","shell.execute_reply":"2024-01-13T10:12:38.149905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The EEG data of eeg_id=3449810742 has 14800 samples and the last subsample ends at 14800.","metadata":{}},{"cell_type":"markdown","source":"### inspiration:\n- split off and save subsample to augment our training data.\n- pseudo labeling by split off more unannotated data.\n\nIf there is anything wrong, please leave your comment and correct me.","metadata":{}}]}