{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from fastai.vision.all import *\nfrom fastcore.parallel import *\n\npath = Path('/kaggle/input/hms-harmful-brain-activity-classification')\n\npath.ls()","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:35:18.557061Z","iopub.execute_input":"2024-04-05T19:35:18.557455Z","iopub.status.idle":"2024-04-05T19:35:31.745690Z","shell.execute_reply.started":"2024-04-05T19:35:18.557422Z","shell.execute_reply":"2024-04-05T19:35:31.744583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Background","metadata":{}},{"cell_type":"markdown","source":"In this notebook, I'll create a new dataset of spectrograms following the step taken in the [HMS-HBAC: KerasCV Starter Notebook](https://www.kaggle.com/code/awsaf49/hms-hbac-kerascv-starter-notebook#%F0%9F%93%81-|-Dataset-Path) to slice the array to a width of 300 (the minimum number of rows in the spectrogram data).","metadata":{}},{"cell_type":"markdown","source":"## Look at the Data","metadata":{}},{"cell_type":"code","source":"spec_df = pd.read_parquet((path/'train_spectrograms').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:36:45.690012Z","iopub.execute_input":"2024-04-05T19:36:45.690741Z","iopub.status.idle":"2024-04-05T19:36:46.188752Z","shell.execute_reply.started":"2024-04-05T19:36:45.690578Z","shell.execute_reply":"2024-04-05T19:36:46.187423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\n\n# replace NA with 0\ndata = spec_df.fillna(0)\n\n# convert DataFrame to array\ndata = data.values[:, 1:]\n    \n# transpose\ndata = data.T\ndata = data.astype(\"float32\")\n\n# crop to 300 values\ndata = data[:, :300]\n\n# convert array to PILImage\nim = PILImage.create(Image.fromarray((data * 255).astype(np.uint8)))\nim","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:36:52.633815Z","iopub.execute_input":"2024-04-05T19:36:52.634199Z","iopub.status.idle":"2024-04-05T19:36:52.678359Z","shell.execute_reply.started":"2024-04-05T19:36:52.634167Z","shell.execute_reply":"2024-04-05T19:36:52.677473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:36:57.101450Z","iopub.execute_input":"2024-04-05T19:36:57.101840Z","iopub.status.idle":"2024-04-05T19:36:57.108219Z","shell.execute_reply.started":"2024-04-05T19:36:57.101813Z","shell.execute_reply":"2024-04-05T19:36:57.107173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Targets","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(path/'train.csv')\n    \ncols = ['spectrogram_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nagg_funcs = {c: 'sum' for c in cols if 'vote' in c}\n\nunique_df = df[cols].groupby(['spectrogram_id'], as_index=False).agg(agg_funcs)\nunique_df['target'] = unique_df[[c for c in cols if 'vote' in c]].idxmax(axis=1)\n\nunique_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:37:00.274404Z","iopub.execute_input":"2024-04-05T19:37:00.274812Z","iopub.status.idle":"2024-04-05T19:37:00.622541Z","shell.execute_reply.started":"2024-04-05T19:37:00.274774Z","shell.execute_reply":"2024-04-05T19:37:00.621403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:37:01.277887Z","iopub.execute_input":"2024-04-05T19:37:01.279231Z","iopub.status.idle":"2024-04-05T19:37:01.291013Z","shell.execute_reply.started":"2024-04-05T19:37:01.279179Z","shell.execute_reply":"2024-04-05T19:37:01.289531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Output Directories","metadata":{}},{"cell_type":"code","source":"# create output folders to hold spectrograms\nSPEC_DIR = \"/kaggle/working\"\nos.makedirs(SPEC_DIR+'/train_spectrograms', exist_ok=True)\n\nfor targ in ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']:\n    os.makedirs(SPEC_DIR+'/train_spectrograms'+'/'+targ, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:37:06.657873Z","iopub.execute_input":"2024-04-05T19:37:06.658486Z","iopub.status.idle":"2024-04-05T19:37:06.664171Z","shell.execute_reply.started":"2024-04-05T19:37:06.658455Z","shell.execute_reply":"2024-04-05T19:37:06.663077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Path(\"/kaggle/working/train_spectrograms\").ls()","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:37:08.668836Z","iopub.execute_input":"2024-04-05T19:37:08.669212Z","iopub.status.idle":"2024-04-05T19:37:08.677887Z","shell.execute_reply.started":"2024-04-05T19:37:08.669185Z","shell.execute_reply":"2024-04-05T19:37:08.676612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate All Images","metadata":{}},{"cell_type":"code","source":"def process_spec(spec_id, split=\"train\"):\n    # read the data\n    data = pd.read_parquet(path/f'{split}_spectrograms'/f'{spec_id}.parquet')\n    \n    # read the label\n    label = unique_df[unique_df.spectrogram_id == spec_id][\"target\"].item()\n    \n    # replace NA with 0\n    data = data.fillna(0)\n    \n    # convert DataFrame to array\n    data = data.values[:, 1:]\n    \n    # transpose\n    data = data.T\n    data = data.astype(\"float32\")\n    \n    # crop to 300 values\n    data = data[:, :300]\n    \n    # convert array to PILImage\n    im = PILImage.create(Image.fromarray((data * 255).astype(np.uint8)))\n    im.save(f\"{SPEC_DIR}/{split}_spectrograms/{label}/{spec_id}.png\")","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:37:39.697503Z","iopub.execute_input":"2024-04-05T19:37:39.697926Z","iopub.status.idle":"2024-04-05T19:37:39.705282Z","shell.execute_reply.started":"2024-04-05T19:37:39.697898Z","shell.execute_reply":"2024-04-05T19:37:39.703906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spec_ids = df[\"spectrogram_id\"].unique()\nlen(spec_ids)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:37:40.332827Z","iopub.execute_input":"2024-04-05T19:37:40.333876Z","iopub.status.idle":"2024-04-05T19:37:40.342680Z","shell.execute_reply.started":"2024-04-05T19:37:40.333829Z","shell.execute_reply":"2024-04-05T19:37:40.341734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"warnings.filterwarnings(\"ignore\")\nparallel(process_spec, spec_ids, split='train', n_workers=4)\nwarnings.filterwarnings(\"default\")","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:37:44.493769Z","iopub.execute_input":"2024-04-05T19:37:44.494640Z","iopub.status.idle":"2024-04-05T19:42:21.656937Z","shell.execute_reply.started":"2024-04-05T19:37:44.494592Z","shell.execute_reply":"2024-04-05T19:42:21.654837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll view a few images to make sure they look okay:","metadata":{}},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/lrda_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:44:32.459885Z","iopub.execute_input":"2024-04-05T19:44:32.461284Z","iopub.status.idle":"2024-04-05T19:44:32.528992Z","shell.execute_reply.started":"2024-04-05T19:44:32.461242Z","shell.execute_reply":"2024-04-05T19:44:32.527898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/gpd_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:44:36.613000Z","iopub.execute_input":"2024-04-05T19:44:36.613397Z","iopub.status.idle":"2024-04-05T19:44:36.668219Z","shell.execute_reply.started":"2024-04-05T19:44:36.613369Z","shell.execute_reply":"2024-04-05T19:44:36.667087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/lpd_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:44:39.493345Z","iopub.execute_input":"2024-04-05T19:44:39.493746Z","iopub.status.idle":"2024-04-05T19:44:39.564273Z","shell.execute_reply.started":"2024-04-05T19:44:39.493717Z","shell.execute_reply":"2024-04-05T19:44:39.562841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/other_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:44:42.704826Z","iopub.execute_input":"2024-04-05T19:44:42.705226Z","iopub.status.idle":"2024-04-05T19:44:42.780746Z","shell.execute_reply.started":"2024-04-05T19:44:42.705196Z","shell.execute_reply":"2024-04-05T19:44:42.779771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll spot-check the images to make sure that they are in the correct folders (class labels).","metadata":{}},{"cell_type":"code","source":"for vote in ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']:\n    for fpath in (Path(\"/kaggle/working/train_spectrograms\")/vote).ls()[:3]:\n        print(vote, vote == unique_df[unique_df.spectrogram_id == int(fpath.stem)]['target'].item())","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:44:46.349857Z","iopub.execute_input":"2024-04-05T19:44:46.350232Z","iopub.status.idle":"2024-04-05T19:44:46.393954Z","shell.execute_reply.started":"2024-04-05T19:44:46.350205Z","shell.execute_reply":"2024-04-05T19:44:46.392695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll make sure that I captured all of the images in the DataFrame:","metadata":{}},{"cell_type":"code","source":"files = get_image_files(Path(\"/kaggle/working/train_spectrograms\"))\nlen(files)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:44:50.712553Z","iopub.execute_input":"2024-04-05T19:44:50.712977Z","iopub.status.idle":"2024-04-05T19:44:50.791106Z","shell.execute_reply.started":"2024-04-05T19:44:50.712945Z","shell.execute_reply":"2024-04-05T19:44:50.790070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(unique_df)","metadata":{"execution":{"iopub.status.busy":"2024-04-05T19:44:53.710334Z","iopub.execute_input":"2024-04-05T19:44:53.710789Z","iopub.status.idle":"2024-04-05T19:44:53.718229Z","shell.execute_reply.started":"2024-04-05T19:44:53.710745Z","shell.execute_reply":"2024-04-05T19:44:53.717010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks good! I'll save this notebook version and then convert the output folder to a Kaggle Dataset that I can use for training.","metadata":{}}]}