{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from fastai.vision.all import *\nfrom fastcore.parallel import *\n\npath = Path('/kaggle/input/hms-harmful-brain-activity-classification')\n\npath.ls()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T03:48:04.195485Z","iopub.execute_input":"2024-04-04T03:48:04.195864Z","iopub.status.idle":"2024-04-04T03:48:15.147586Z","shell.execute_reply.started":"2024-04-04T03:48:04.195834Z","shell.execute_reply":"2024-04-04T03:48:15.146707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Background","metadata":{}},{"cell_type":"markdown","source":"In this notebook, I'll create a new dataset of spectrograms following the steps taken in the [HMS-HBAC: KerasCV Starter Notebook](https://www.kaggle.com/code/awsaf49/hms-hbac-kerascv-starter-notebook#%F0%9F%93%81-|-Dataset-Path). Namely:\n\n- Clip the spectrogram data to avoid 0s\n- Take the log of spectrogram data to emphasize differences \n- Normalize the spectrogram data\n- **Slice the array to a width of 300 (the minimum number of rows in the spectrogram data)** <-- new ","metadata":{}},{"cell_type":"markdown","source":"## Look at the Data","metadata":{}},{"cell_type":"code","source":"spec_df = pd.read_parquet((path/'train_spectrograms').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-04T03:48:56.557898Z","iopub.execute_input":"2024-04-04T03:48:56.558548Z","iopub.status.idle":"2024-04-04T03:48:57.094120Z","shell.execute_reply.started":"2024-04-04T03:48:56.558517Z","shell.execute_reply":"2024-04-04T03:48:57.093248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\n\n# replace NA with 0\ndata = spec_df.fillna(0)\n\n# convert DataFrame to array\ndata = data.values[:, 1:]\n    \n# transpose\ndata = data.T\ndata = data.astype(\"float32\")\n\n# clip data to avoid 0s\ndata = np.clip(data, math.exp(-4), math.exp(8))\n\n# take log data to magnify differences\ndata = np.log(data)\n\n# normalize data\ndata=(data-data.mean())/data.std() + 1e-6\n\n# convert to 3 channels\ndata = np.tile(data[..., None], (1, 1, 3))\n\n# crop to 300 values\ndata = data[:, :300, :]\n\n# convert array to PILImage\nim = PILImage.create(Image.fromarray((data * 255).astype(np.uint8)))\nim","metadata":{"execution":{"iopub.status.busy":"2024-04-04T03:50:13.972063Z","iopub.execute_input":"2024-04-04T03:50:13.973088Z","iopub.status.idle":"2024-04-04T03:50:14.037433Z","shell.execute_reply.started":"2024-04-04T03:50:13.973053Z","shell.execute_reply":"2024-04-04T03:50:14.036480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-04T03:50:20.839498Z","iopub.execute_input":"2024-04-04T03:50:20.840569Z","iopub.status.idle":"2024-04-04T03:50:20.847989Z","shell.execute_reply.started":"2024-04-04T03:50:20.840524Z","shell.execute_reply":"2024-04-04T03:50:20.846553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Targets","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(path/'train.csv')\n    \ncols = ['spectrogram_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nagg_funcs = {c: 'sum' for c in cols if 'vote' in c}\n\nunique_df = df[cols].groupby(['spectrogram_id'], as_index=False).agg(agg_funcs)\nunique_df['target'] = unique_df[[c for c in cols if 'vote' in c]].idxmax(axis=1)\n\nunique_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T03:50:27.425860Z","iopub.execute_input":"2024-04-04T03:50:27.426602Z","iopub.status.idle":"2024-04-04T03:50:27.692135Z","shell.execute_reply.started":"2024-04-04T03:50:27.426565Z","shell.execute_reply":"2024-04-04T03:50:27.691135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T03:50:34.024452Z","iopub.execute_input":"2024-04-04T03:50:34.024861Z","iopub.status.idle":"2024-04-04T03:50:34.034513Z","shell.execute_reply.started":"2024-04-04T03:50:34.024830Z","shell.execute_reply":"2024-04-04T03:50:34.033189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Output Directories","metadata":{}},{"cell_type":"code","source":"# create output folders to hold spectrograms\nSPEC_DIR = \"/kaggle/working\"\nos.makedirs(SPEC_DIR+'/train_spectrograms', exist_ok=True)\n\nfor targ in ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']:\n    os.makedirs(SPEC_DIR+'/train_spectrograms'+'/'+targ, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T03:50:48.135918Z","iopub.execute_input":"2024-04-04T03:50:48.136345Z","iopub.status.idle":"2024-04-04T03:50:48.142552Z","shell.execute_reply.started":"2024-04-04T03:50:48.136314Z","shell.execute_reply":"2024-04-04T03:50:48.141363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Path(\"/kaggle/working/train_spectrograms\").ls()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T03:50:49.707571Z","iopub.execute_input":"2024-04-04T03:50:49.707980Z","iopub.status.idle":"2024-04-04T03:50:49.714666Z","shell.execute_reply.started":"2024-04-04T03:50:49.707948Z","shell.execute_reply":"2024-04-04T03:50:49.713602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate All Images","metadata":{}},{"cell_type":"code","source":"def process_spec(spec_id, split=\"train\"):\n    # read the data\n    data = pd.read_parquet(path/f'{split}_spectrograms'/f'{spec_id}.parquet')\n    \n    # read the label\n    label = unique_df[unique_df.spectrogram_id == spec_id][\"target\"].item()\n    \n    # replace NA with 0\n    data = data.fillna(0)\n    \n    # convert DataFrame to array\n    data = data.values[:, 1:]\n    \n    # transpose\n    data = data.T\n    data = data.astype(\"float32\")\n    \n    # clip data to avoid 0s\n    data = np.clip(data, math.exp(-4), math.exp(8))\n\n    # take log data to magnify differences\n    data = np.log(data)\n\n    # normalize data\n    data=(data-data.mean())/data.std() + 1e-6\n\n    # convert to 3 channels\n    data = np.tile(data[..., None], (1, 1, 3))\n    \n    # crop to 300 values\n    data = data[:, :300, :]\n    \n    # convert array to PILImage\n    im = PILImage.create(Image.fromarray((data * 255).astype(np.uint8)))\n    im.save(f\"{SPEC_DIR}/{split}_spectrograms/{label}/{spec_id}.png\")","metadata":{"execution":{"iopub.status.busy":"2024-04-04T03:51:08.732365Z","iopub.execute_input":"2024-04-04T03:51:08.732770Z","iopub.status.idle":"2024-04-04T03:51:08.740872Z","shell.execute_reply.started":"2024-04-04T03:51:08.732739Z","shell.execute_reply":"2024-04-04T03:51:08.739601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spec_ids = df[\"spectrogram_id\"].unique()\nlen(spec_ids)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T03:51:10.333712Z","iopub.execute_input":"2024-04-04T03:51:10.334126Z","iopub.status.idle":"2024-04-04T03:51:10.344145Z","shell.execute_reply.started":"2024-04-04T03:51:10.334095Z","shell.execute_reply":"2024-04-04T03:51:10.342977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"warnings.filterwarnings(\"ignore\")\nparallel(process_spec, spec_ids, split='train', n_workers=4)\nwarnings.filterwarnings(\"default\")","metadata":{"execution":{"iopub.status.busy":"2024-04-04T03:51:15.095319Z","iopub.execute_input":"2024-04-04T03:51:15.095722Z","iopub.status.idle":"2024-04-04T03:58:04.675678Z","shell.execute_reply.started":"2024-04-04T03:51:15.095693Z","shell.execute_reply":"2024-04-04T03:58:04.673880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll view a few images to make sure they look okay:","metadata":{}},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/lrda_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-04T04:01:31.185007Z","iopub.execute_input":"2024-04-04T04:01:31.186068Z","iopub.status.idle":"2024-04-04T04:01:31.275628Z","shell.execute_reply.started":"2024-04-04T04:01:31.186032Z","shell.execute_reply":"2024-04-04T04:01:31.274536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/gpd_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-04T04:01:33.961180Z","iopub.execute_input":"2024-04-04T04:01:33.961609Z","iopub.status.idle":"2024-04-04T04:01:34.010010Z","shell.execute_reply.started":"2024-04-04T04:01:33.961581Z","shell.execute_reply":"2024-04-04T04:01:34.008924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/lpd_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-04T04:01:36.927333Z","iopub.execute_input":"2024-04-04T04:01:36.927754Z","iopub.status.idle":"2024-04-04T04:01:37.002714Z","shell.execute_reply.started":"2024-04-04T04:01:36.927725Z","shell.execute_reply":"2024-04-04T04:01:37.001856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/other_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-04T04:01:39.688996Z","iopub.execute_input":"2024-04-04T04:01:39.689434Z","iopub.status.idle":"2024-04-04T04:01:39.761903Z","shell.execute_reply.started":"2024-04-04T04:01:39.689401Z","shell.execute_reply":"2024-04-04T04:01:39.760753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll spot-check the images to make sure that they are in the correct folders (class labels).","metadata":{}},{"cell_type":"code","source":"for vote in ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']:\n    for fpath in (Path(\"/kaggle/working/train_spectrograms\")/vote).ls()[:3]:\n        print(vote, vote == unique_df[unique_df.spectrogram_id == int(fpath.stem)]['target'].item())","metadata":{"execution":{"iopub.status.busy":"2024-04-04T04:01:43.232155Z","iopub.execute_input":"2024-04-04T04:01:43.232542Z","iopub.status.idle":"2024-04-04T04:01:43.278035Z","shell.execute_reply.started":"2024-04-04T04:01:43.232513Z","shell.execute_reply":"2024-04-04T04:01:43.276930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll make sure that I captured all of the images in the DataFrame:","metadata":{}},{"cell_type":"code","source":"files = get_image_files(Path(\"/kaggle/working/train_spectrograms\"))\nlen(files)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T04:01:45.965071Z","iopub.execute_input":"2024-04-04T04:01:45.965475Z","iopub.status.idle":"2024-04-04T04:01:46.387088Z","shell.execute_reply.started":"2024-04-04T04:01:45.965446Z","shell.execute_reply":"2024-04-04T04:01:46.386063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(unique_df)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T04:01:47.319566Z","iopub.execute_input":"2024-04-04T04:01:47.320687Z","iopub.status.idle":"2024-04-04T04:01:47.326809Z","shell.execute_reply.started":"2024-04-04T04:01:47.320650Z","shell.execute_reply":"2024-04-04T04:01:47.325627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks good! I'll save this notebook version and then convert the output folder to a Kaggle Dataset that I can use for training.","metadata":{}}]}