{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30746,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-15T08:22:22.718805Z","iopub.execute_input":"2024-07-15T08:22:22.719791Z","iopub.status.idle":"2024-07-15T08:22:22.728799Z","shell.execute_reply.started":"2024-07-15T08:22:22.719739Z","shell.execute_reply":"2024-07-15T08:22:22.727294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport torch","metadata":{"execution":{"iopub.status.busy":"2024-07-15T09:45:05.851128Z","iopub.execute_input":"2024-07-15T09:45:05.852213Z","iopub.status.idle":"2024-07-15T09:45:09.212784Z","shell.execute_reply.started":"2024-07-15T09:45:05.852174Z","shell.execute_reply":"2024-07-15T09:45:09.211571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#some helper function from starter notebook\nclass CFG:\n    verbose = 1  # Verbosity\n    seed = 42  # Random seed\n    preset = \"efficientnetv2_b2_imagenet\"  # Name of pretrained classifier\n    image_size = [400, 300]  # Input image size\n    epochs = 13 # Training epochs\n    batch_size = 64  # Batch size\n    lr_mode = \"cos\" # LR scheduler mode from one of \"cos\", \"step\", \"exp\"\n    drop_remainder = True  # Drop incomplete batches\n    num_classes = 6 # Number of classes in the dataset\n    fold = 0 # Which fold to set as validation data\n    class_names = ['Seizure', 'LPD', 'GPD', 'LRDA','GRDA', 'Other']\n    label2name = dict(enumerate(class_names))\n    name2label = {v:k for k, v in label2name.items()}","metadata":{"execution":{"iopub.status.busy":"2024-07-15T08:39:55.677248Z","iopub.execute_input":"2024-07-15T08:39:55.677659Z","iopub.status.idle":"2024-07-15T08:39:55.685538Z","shell.execute_reply.started":"2024-07-15T08:39:55.677629Z","shell.execute_reply":"2024-07-15T08:39:55.684262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification\"\n\nSPEC_DIR = \"/tmp/dataset/hms-hbac\"\nos.makedirs(SPEC_DIR+'/train_spectrograms', exist_ok=True)\nos.makedirs(SPEC_DIR+'/test_spectrograms', exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T08:33:39.07032Z","iopub.execute_input":"2024-07-15T08:33:39.070767Z","iopub.status.idle":"2024-07-15T08:33:39.07829Z","shell.execute_reply.started":"2024-07-15T08:33:39.070733Z","shell.execute_reply":"2024-07-15T08:33:39.076796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#load file with data description\ndf = pd.read_csv(f'{BASE_PATH}/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-15T08:34:50.401972Z","iopub.execute_input":"2024-07-15T08:34:50.403371Z","iopub.status.idle":"2024-07-15T08:34:50.629031Z","shell.execute_reply.started":"2024-07-15T08:34:50.403327Z","shell.execute_reply":"2024-07-15T08:34:50.627536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#number of classes\ndf['expert_consensus'].value_counts()\n#patient_id -> ? to avoid splitting spectrograms from the same patients into train and test","metadata":{"execution":{"iopub.status.busy":"2024-07-15T08:35:01.635992Z","iopub.execute_input":"2024-07-15T08:35:01.636382Z","iopub.status.idle":"2024-07-15T08:35:01.669913Z","shell.execute_reply.started":"2024-07-15T08:35:01.636353Z","shell.execute_reply":"2024-07-15T08:35:01.668449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['eeg_path'] = f'{BASE_PATH}/train_eegs/'+df['eeg_id'].astype(str)+'.parquet'\ndf['spec_path'] = f'{BASE_PATH}/train_spectrograms/'+df['spectrogram_id'].astype(str)+'.parquet'\ndf['spec2_path'] = f'{SPEC_DIR}/train_spectrograms/'+df['spectrogram_id'].astype(str)+'.npy'\ndf['class_name'] = df.expert_consensus.copy()\ndf['class_label'] = df.expert_consensus.map(CFG.name2label)\ndisplay(df.head(2))","metadata":{"execution":{"iopub.status.busy":"2024-07-15T08:39:59.903863Z","iopub.execute_input":"2024-07-15T08:39:59.904302Z","iopub.status.idle":"2024-07-15T08:40:00.274384Z","shell.execute_reply.started":"2024-07-15T08:39:59.904267Z","shell.execute_reply":"2024-07-15T08:40:00.273023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#function converting parqour files to npy arrays and saving them\ndef process_spec(spec_id, split=\"train\"):\n    spec_path = f\"{BASE_PATH}/{split}_spectrograms/{spec_id}.parquet\"\n    spec = pd.read_parquet(spec_path)\n    spec = spec.fillna(0).values[:, 1:].T # fill NaN values with 0, transpose for (Time, Freq) -> (Freq, Time)\n    spec = spec.astype(\"float32\")\n    np.save(f\"{SPEC_DIR}/{split}_spectrograms/{spec_id}.npy\", spec)\n\n# Get unique spec_ids of train and valid data\nspec_ids = df[\"spectrogram_id\"].unique()","metadata":{"execution":{"iopub.status.busy":"2024-07-15T09:28:02.367692Z","iopub.execute_input":"2024-07-15T09:28:02.368128Z","iopub.status.idle":"2024-07-15T09:28:02.377007Z","shell.execute_reply.started":"2024-07-15T09:28:02.368094Z","shell.execute_reply":"2024-07-15T09:28:02.375814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_specs = dict()\nmin_time_shape = 500\n\nfor idd in range(0, 5): #len(spec_ids)):\n    process_spec(spec_ids[idd])\n    all_specs[spec_ids[idd]] = np.load(str(spec_ids[idd]) + '.npy')\n    time_sh = np.shape(all_specs[spec_ids[idd]])[1]\n    if time_sh < min_time_shape:\n        min_time_shape = time_sh\n    #print(np.shape(all_specs[spec_ids[idd]]))\nos.listdir(SPEC_DIR+'/train_spectrograms')\nprint(min_time_shape)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T09:38:30.308308Z","iopub.execute_input":"2024-07-15T09:38:30.308711Z","iopub.status.idle":"2024-07-15T09:38:30.495386Z","shell.execute_reply.started":"2024-07-15T09:38:30.308682Z","shell.execute_reply":"2024-07-15T09:38:30.493842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.chdir(SPEC_DIR+'/train_spectrograms')","metadata":{"execution":{"iopub.status.busy":"2024-07-15T08:46:59.171404Z","iopub.execute_input":"2024-07-15T08:46:59.171847Z","iopub.status.idle":"2024-07-15T08:46:59.177302Z","shell.execute_reply.started":"2024-07-15T08:46:59.171815Z","shell.execute_reply":"2024-07-15T08:46:59.176101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spec3 = np.load(str(spec_ids[2]) + '.npy')","metadata":{"execution":{"iopub.status.busy":"2024-07-15T09:18:34.131191Z","iopub.execute_input":"2024-07-15T09:18:34.131655Z","iopub.status.idle":"2024-07-15T09:18:34.138349Z","shell.execute_reply.started":"2024-07-15T09:18:34.131615Z","shell.execute_reply":"2024-07-15T09:18:34.137243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.shape(spec1)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T09:09:12.335726Z","iopub.execute_input":"2024-07-15T09:09:12.336185Z","iopub.status.idle":"2024-07-15T09:09:12.345166Z","shell.execute_reply.started":"2024-07-15T09:09:12.33615Z","shell.execute_reply":"2024-07-15T09:09:12.343929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_specs.keys()","metadata":{"execution":{"iopub.status.busy":"2024-07-15T09:23:27.241036Z","iopub.execute_input":"2024-07-15T09:23:27.241493Z","iopub.status.idle":"2024-07-15T09:23:27.248971Z","shell.execute_reply.started":"2024-07-15T09:23:27.241461Z","shell.execute_reply":"2024-07-15T09:23:27.247576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(all_specs[spec_ids[3]].T)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T09:33:18.216271Z","iopub.execute_input":"2024-07-15T09:33:18.216676Z","iopub.status.idle":"2024-07-15T09:33:19.183556Z","shell.execute_reply.started":"2024-07-15T09:33:18.216644Z","shell.execute_reply":"2024-07-15T09:33:19.182396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#transform data to torch tensor","metadata":{}},{"cell_type":"code","source":"spec2 = np.load(str(spec_ids[2]) + '.npy')\nspec2t = torch.Tensor(spec2)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T09:48:09.870651Z","iopub.execute_input":"2024-07-15T09:48:09.871101Z","iopub.status.idle":"2024-07-15T09:48:09.879909Z","shell.execute_reply.started":"2024-07-15T09:48:09.87105Z","shell.execute_reply":"2024-07-15T09:48:09.877222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#do torch transformations like resize, scaling, change color space etc (but such that not change the info in data, not rotation or cropp)\nfrom torchvision.transforms import v2\n\n#just some random transforms to see if it works at all, later choose some that makes sense\ntransforms = v2.Compose([\n    v2.RandomHorizontalFlip(p=0.5),\n    v2.ToDtype(torch.float32, scale=True)#,\n    #v2.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\ntransformed_spectrogram = transforms(spec2t)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T09:58:50.89517Z","iopub.execute_input":"2024-07-15T09:58:50.895731Z","iopub.status.idle":"2024-07-15T09:58:52.916841Z","shell.execute_reply.started":"2024-07-15T09:58:50.895688Z","shell.execute_reply":"2024-07-15T09:58:52.915376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(spec2t)\nprint(transformed_spectrogram)","metadata":{"execution":{"iopub.status.busy":"2024-07-15T09:59:37.368093Z","iopub.execute_input":"2024-07-15T09:59:37.368541Z","iopub.status.idle":"2024-07-15T09:59:37.379799Z","shell.execute_reply.started":"2024-07-15T09:59:37.368505Z","shell.execute_reply":"2024-07-15T09:59:37.378432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"TO DO:\n - DATA LOADER FROM PYTORCH FUNCTIONS\n - TRANSFORMATIONS OF DATA\n - PLOT A SAMPLE SPECTRORAM BEFORE AND AFTER TRANSFORMS","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}