{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from fastai.vision.all import *\nfrom fastcore.parallel import *\n\npath = Path('/kaggle/input/hms-harmful-brain-activity-classification')\n\npath.ls()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T00:27:54.643875Z","iopub.execute_input":"2024-04-03T00:27:54.644222Z","iopub.status.idle":"2024-04-03T00:28:06.422120Z","shell.execute_reply.started":"2024-04-03T00:27:54.644187Z","shell.execute_reply":"2024-04-03T00:28:06.420949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Background","metadata":{}},{"cell_type":"markdown","source":"In this notebook, I'll create a new dataset of spectrograms following the steps taken in the [HMS-HBAC: KerasCV Starter Notebook](https://www.kaggle.com/code/awsaf49/hms-hbac-kerascv-starter-notebook#%F0%9F%93%81-|-Dataset-Path). Namely:\n\n- Clip the spectrogram data to avoid 0s\n- Take the log of spectrogram data to emphasize differences \n- Normalize the spectrogram data","metadata":{}},{"cell_type":"markdown","source":"## Look at the Data","metadata":{}},{"cell_type":"code","source":"spec_df = pd.read_parquet((path/'train_spectrograms').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:08:40.658305Z","iopub.execute_input":"2024-04-03T01:08:40.659153Z","iopub.status.idle":"2024-04-03T01:08:40.718852Z","shell.execute_reply.started":"2024-04-03T01:08:40.659115Z","shell.execute_reply":"2024-04-03T01:08:40.717807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\n\n# replace NA with 0\ndata = spec_df.fillna(0)\n\n# convert DataFrame to array\ndata = data.values[:, 1:]\n    \n# transpose\ndata = data.T\ndata = data.astype(\"float32\")\n\n# clip data to avoid 0s\ndata = np.clip(data, math.exp(-4), math.exp(8))\n\n# take log data to magnify differences\ndata = np.log(data)\n\n# normalize data\ndata=(data-data.mean())/data.std() + 1e-6\n\n# convert to 3 channels\ndata = np.tile(data[..., None], (1, 1, 3))\n\n# convert array to PILImage\nim = PILImage.create(Image.fromarray((data * 255).astype(np.uint8)))\nim","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:08:52.414369Z","iopub.execute_input":"2024-04-03T01:08:52.417114Z","iopub.status.idle":"2024-04-03T01:08:52.503582Z","shell.execute_reply.started":"2024-04-03T01:08:52.417072Z","shell.execute_reply":"2024-04-03T01:08:52.502560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:09:24.098722Z","iopub.execute_input":"2024-04-03T01:09:24.099148Z","iopub.status.idle":"2024-04-03T01:09:24.106378Z","shell.execute_reply.started":"2024-04-03T01:09:24.099115Z","shell.execute_reply":"2024-04-03T01:09:24.105251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Targets","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(path/'train.csv')\n    \ncols = ['spectrogram_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nagg_funcs = {c: 'sum' for c in cols if 'vote' in c}\n\nunique_df = df[cols].groupby(['spectrogram_id'], as_index=False).agg(agg_funcs)\nunique_df['target'] = unique_df[[c for c in cols if 'vote' in c]].idxmax(axis=1)\n\nunique_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:10:08.061386Z","iopub.execute_input":"2024-04-03T01:10:08.062478Z","iopub.status.idle":"2024-04-03T01:10:08.447887Z","shell.execute_reply.started":"2024-04-03T01:10:08.062431Z","shell.execute_reply":"2024-04-03T01:10:08.446777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:10:24.978495Z","iopub.execute_input":"2024-04-03T01:10:24.979248Z","iopub.status.idle":"2024-04-03T01:10:24.989331Z","shell.execute_reply.started":"2024-04-03T01:10:24.979210Z","shell.execute_reply":"2024-04-03T01:10:24.988319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Output Directories","metadata":{}},{"cell_type":"code","source":"# create output folders to hold spectrograms\nSPEC_DIR = \"/kaggle/working\"\nos.makedirs(SPEC_DIR+'/train_spectrograms', exist_ok=True)\n\nfor targ in ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']:\n    os.makedirs(SPEC_DIR+'/train_spectrograms'+'/'+targ, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:11:07.070309Z","iopub.execute_input":"2024-04-03T01:11:07.070711Z","iopub.status.idle":"2024-04-03T01:11:07.077552Z","shell.execute_reply.started":"2024-04-03T01:11:07.070683Z","shell.execute_reply":"2024-04-03T01:11:07.076243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Path(\"/kaggle/working/train_spectrograms\").ls()","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:11:33.078823Z","iopub.execute_input":"2024-04-03T01:11:33.079235Z","iopub.status.idle":"2024-04-03T01:11:33.086236Z","shell.execute_reply.started":"2024-04-03T01:11:33.079202Z","shell.execute_reply":"2024-04-03T01:11:33.085014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate All Images","metadata":{}},{"cell_type":"code","source":"def process_spec(spec_id, split=\"train\"):\n    # read the data\n    data = pd.read_parquet(path/f'{split}_spectrograms'/f'{spec_id}.parquet')\n    \n    # read the label\n    label = unique_df[unique_df.spectrogram_id == spec_id][\"target\"].item()\n    \n    # replace NA with 0\n    data = data.fillna(0)\n    \n    # convert DataFrame to array\n    data = data.values[:, 1:]\n    \n    # transpose\n    data = data.T\n    data = data.astype(\"float32\")\n    \n    # clip data to avoid 0s\n    data = np.clip(data, math.exp(-4), math.exp(8))\n\n    # take log data to magnify differences\n    data = np.log(data)\n\n    # normalize data\n    data=(data-data.mean())/data.std() + 1e-6\n\n    # convert to 3 channels\n    data = np.tile(data[..., None], (1, 1, 3))\n    \n    # convert array to PILImage\n    im = PILImage.create(Image.fromarray((data * 255).astype(np.uint8)))\n    im.save(f\"{SPEC_DIR}/{split}_spectrograms/{label}/{spec_id}.png\")","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:12:48.699247Z","iopub.execute_input":"2024-04-03T01:12:48.699626Z","iopub.status.idle":"2024-04-03T01:12:48.708047Z","shell.execute_reply.started":"2024-04-03T01:12:48.699600Z","shell.execute_reply":"2024-04-03T01:12:48.706818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spec_ids = df[\"spectrogram_id\"].unique()\nlen(spec_ids)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:12:56.682833Z","iopub.execute_input":"2024-04-03T01:12:56.683198Z","iopub.status.idle":"2024-04-03T01:12:56.693329Z","shell.execute_reply.started":"2024-04-03T01:12:56.683172Z","shell.execute_reply":"2024-04-03T01:12:56.692313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"warnings.filterwarnings(\"ignore\")\nparallel(process_spec, spec_ids, split='train', n_workers=4)\nwarnings.filterwarnings(\"default\")","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:13:20.123396Z","iopub.execute_input":"2024-04-03T01:13:20.123784Z","iopub.status.idle":"2024-04-03T01:20:59.162977Z","shell.execute_reply.started":"2024-04-03T01:13:20.123736Z","shell.execute_reply":"2024-04-03T01:20:59.160724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll view a few images to make sure they look okay:","metadata":{}},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/lrda_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:29:12.710482Z","iopub.execute_input":"2024-04-03T01:29:12.711686Z","iopub.status.idle":"2024-04-03T01:29:12.793293Z","shell.execute_reply.started":"2024-04-03T01:29:12.711640Z","shell.execute_reply":"2024-04-03T01:29:12.788887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/gpd_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:29:54.714856Z","iopub.execute_input":"2024-04-03T01:29:54.715259Z","iopub.status.idle":"2024-04-03T01:29:54.800360Z","shell.execute_reply.started":"2024-04-03T01:29:54.715229Z","shell.execute_reply":"2024-04-03T01:29:54.799224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/lpd_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:30:01.934688Z","iopub.execute_input":"2024-04-03T01:30:01.935114Z","iopub.status.idle":"2024-04-03T01:30:02.052109Z","shell.execute_reply.started":"2024-04-03T01:30:01.935084Z","shell.execute_reply":"2024-04-03T01:30:02.051020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/other_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:30:19.270692Z","iopub.execute_input":"2024-04-03T01:30:19.271104Z","iopub.status.idle":"2024-04-03T01:30:19.351400Z","shell.execute_reply.started":"2024-04-03T01:30:19.271075Z","shell.execute_reply":"2024-04-03T01:30:19.350196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll spot-check the images to make sure that they are in the correct folders (class labels).","metadata":{}},{"cell_type":"code","source":"for vote in ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']:\n    for fpath in (Path(\"/kaggle/working/train_spectrograms\")/vote).ls()[:3]:\n        print(vote, vote == unique_df[unique_df.spectrogram_id == int(fpath.stem)]['target'].item())","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:26:46.927976Z","iopub.execute_input":"2024-04-03T01:26:46.928456Z","iopub.status.idle":"2024-04-03T01:26:46.986581Z","shell.execute_reply.started":"2024-04-03T01:26:46.928425Z","shell.execute_reply":"2024-04-03T01:26:46.985553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll make sure that I captured all of the images in the DataFrame:","metadata":{}},{"cell_type":"code","source":"files = get_image_files(Path(\"/kaggle/working/train_spectrograms\"))\nlen(files)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:26:49.769326Z","iopub.execute_input":"2024-04-03T01:26:49.770759Z","iopub.status.idle":"2024-04-03T01:26:49.854544Z","shell.execute_reply.started":"2024-04-03T01:26:49.770688Z","shell.execute_reply":"2024-04-03T01:26:49.853265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(unique_df)","metadata":{"execution":{"iopub.status.busy":"2024-04-03T01:26:50.449501Z","iopub.execute_input":"2024-04-03T01:26:50.450517Z","iopub.status.idle":"2024-04-03T01:26:50.458355Z","shell.execute_reply.started":"2024-04-03T01:26:50.450475Z","shell.execute_reply":"2024-04-03T01:26:50.456845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks good! I'll save this notebook version and then convert the output folder to a Kaggle Dataset that I can use for training.","metadata":{}}]}