{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from fastai.vision.all import *\nfrom fastcore.parallel import *\n\npath = Path('/kaggle/input/hms-harmful-brain-activity-classification')\n\npath.ls()","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:24:15.433780Z","iopub.execute_input":"2024-10-09T17:24:15.434103Z","iopub.status.idle":"2024-10-09T17:24:26.705854Z","shell.execute_reply.started":"2024-10-09T17:24:15.434068Z","shell.execute_reply":"2024-10-09T17:24:26.704952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Background","metadata":{}},{"cell_type":"markdown","source":"In this notebook, I'll create a new dataset of spectrograms following the step taken in the [HMS-HBAC: KerasCV Starter Notebook](https://www.kaggle.com/code/awsaf49/hms-hbac-kerascv-starter-notebook#%F0%9F%93%81-|-Dataset-Path) to slice the array to a width of 300 (the minimum number of rows in the spectrogram data).","metadata":{}},{"cell_type":"markdown","source":"## Look at the Data","metadata":{}},{"cell_type":"code","source":"spec_df = pd.read_parquet((path/'train_spectrograms').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:24:26.707673Z","iopub.execute_input":"2024-10-09T17:24:26.708303Z","iopub.status.idle":"2024-10-09T17:24:27.475340Z","shell.execute_reply.started":"2024-10-09T17:24:26.708270Z","shell.execute_reply":"2024-10-09T17:24:27.474520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\n\n# replace NA with 0\ndata = spec_df.fillna(0)\n\n# convert DataFrame to array\ndata = data.values[:, 1:]\n    \n# transpose\ndata = data.T\ndata = data.astype(\"float32\")\n\n# crop to 300 values\ndata = data[:, :300]\n\n# convert array to PILImage\nim = PILImage.create(Image.fromarray((data * 255).astype(np.uint8)))\nim","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:24:27.476468Z","iopub.execute_input":"2024-10-09T17:24:27.476758Z","iopub.status.idle":"2024-10-09T17:24:27.511086Z","shell.execute_reply.started":"2024-10-09T17:24:27.476734Z","shell.execute_reply":"2024-10-09T17:24:27.510224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:24:27.513080Z","iopub.execute_input":"2024-10-09T17:24:27.513346Z","iopub.status.idle":"2024-10-09T17:24:27.519075Z","shell.execute_reply.started":"2024-10-09T17:24:27.513323Z","shell.execute_reply":"2024-10-09T17:24:27.518136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Targets","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(path/'train.csv')\n    \ncols = ['spectrogram_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nagg_funcs = {c: 'sum' for c in cols if 'vote' in c}\n\nunique_df = df[cols].groupby(['spectrogram_id'], as_index=False).agg(agg_funcs)\nunique_df['target'] = unique_df[[c for c in cols if 'vote' in c]].idxmax(axis=1)\n\nunique_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:24:27.520297Z","iopub.execute_input":"2024-10-09T17:24:27.520543Z","iopub.status.idle":"2024-10-09T17:24:27.802811Z","shell.execute_reply.started":"2024-10-09T17:24:27.520522Z","shell.execute_reply":"2024-10-09T17:24:27.801770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:24:27.803875Z","iopub.execute_input":"2024-10-09T17:24:27.804163Z","iopub.status.idle":"2024-10-09T17:24:27.813482Z","shell.execute_reply.started":"2024-10-09T17:24:27.804139Z","shell.execute_reply":"2024-10-09T17:24:27.812577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare Output Directories","metadata":{}},{"cell_type":"code","source":"# create output folders to hold spectrograms\nSPEC_DIR = \"/kaggle/working\"\nos.makedirs(SPEC_DIR+'/train_spectrograms', exist_ok=True)\n\nfor targ in ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']:\n    os.makedirs(SPEC_DIR+'/train_spectrograms'+'/'+targ, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:24:27.814605Z","iopub.execute_input":"2024-10-09T17:24:27.816605Z","iopub.status.idle":"2024-10-09T17:24:27.822700Z","shell.execute_reply.started":"2024-10-09T17:24:27.816580Z","shell.execute_reply":"2024-10-09T17:24:27.821877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Path(\"/kaggle/working/train_spectrograms\").ls()","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:24:27.823862Z","iopub.execute_input":"2024-10-09T17:24:27.824449Z","iopub.status.idle":"2024-10-09T17:24:27.832678Z","shell.execute_reply.started":"2024-10-09T17:24:27.824417Z","shell.execute_reply":"2024-10-09T17:24:27.831742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Generate All Images","metadata":{}},{"cell_type":"code","source":"def process_spec(spec_id, split=\"train\"):\n    # read the data\n    data = pd.read_parquet(path/f'{split}_spectrograms'/f'{spec_id}.parquet')\n    \n    # read the label\n    label = unique_df[unique_df.spectrogram_id == spec_id][\"target\"].item()\n    \n    # replace NA with 0\n    data = data.fillna(0)\n    \n    # convert DataFrame to array\n    data = data.values[:, 1:]\n    \n    # transpose\n    data = data.T\n    data = data.astype(\"float32\")\n    \n    # crop to 300 values\n    data = data[:, :300]\n    \n    # convert array to PILImage\n    im = PILImage.create(Image.fromarray((data * 255).astype(np.uint8)))\n    im.save(f\"{SPEC_DIR}/{split}_spectrograms/{label}/{spec_id}.png\")","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:24:27.833681Z","iopub.execute_input":"2024-10-09T17:24:27.833990Z","iopub.status.idle":"2024-10-09T17:24:27.841456Z","shell.execute_reply.started":"2024-10-09T17:24:27.833965Z","shell.execute_reply":"2024-10-09T17:24:27.840606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spec_ids = df[\"spectrogram_id\"].unique()\nlen(spec_ids)","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:24:27.844403Z","iopub.execute_input":"2024-10-09T17:24:27.844952Z","iopub.status.idle":"2024-10-09T17:24:27.855926Z","shell.execute_reply.started":"2024-10-09T17:24:27.844899Z","shell.execute_reply":"2024-10-09T17:24:27.855052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"warnings.filterwarnings(\"ignore\")\nparallel(process_spec, spec_ids, split='train', n_workers=4)\nwarnings.filterwarnings(\"default\")","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:24:27.856887Z","iopub.execute_input":"2024-10-09T17:24:27.857903Z","iopub.status.idle":"2024-10-09T17:28:23.987532Z","shell.execute_reply.started":"2024-10-09T17:24:27.857880Z","shell.execute_reply":"2024-10-09T17:28:23.986580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll view a few images to make sure they look okay:","metadata":{}},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/lrda_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:28:23.988761Z","iopub.execute_input":"2024-10-09T17:28:23.989112Z","iopub.status.idle":"2024-10-09T17:28:24.041695Z","shell.execute_reply.started":"2024-10-09T17:28:23.989084Z","shell.execute_reply":"2024-10-09T17:28:24.040851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/gpd_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:28:24.042758Z","iopub.execute_input":"2024-10-09T17:28:24.043033Z","iopub.status.idle":"2024-10-09T17:28:24.095717Z","shell.execute_reply.started":"2024-10-09T17:28:24.043009Z","shell.execute_reply":"2024-10-09T17:28:24.094817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/lpd_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:28:24.096706Z","iopub.execute_input":"2024-10-09T17:28:24.096982Z","iopub.status.idle":"2024-10-09T17:28:24.138888Z","shell.execute_reply.started":"2024-10-09T17:28:24.096958Z","shell.execute_reply":"2024-10-09T17:28:24.138054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PILImage.create(Path('/kaggle/working/train_spectrograms/other_vote').ls()[0])","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:28:24.139958Z","iopub.execute_input":"2024-10-09T17:28:24.140219Z","iopub.status.idle":"2024-10-09T17:28:24.203249Z","shell.execute_reply.started":"2024-10-09T17:28:24.140197Z","shell.execute_reply":"2024-10-09T17:28:24.202418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll spot-check the images to make sure that they are in the correct folders (class labels).","metadata":{}},{"cell_type":"code","source":"for vote in ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']:\n    for fpath in (Path(\"/kaggle/working/train_spectrograms\")/vote).ls()[:3]:\n        print(vote, vote == unique_df[unique_df.spectrogram_id == int(fpath.stem)]['target'].item())","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:28:24.204391Z","iopub.execute_input":"2024-10-09T17:28:24.204696Z","iopub.status.idle":"2024-10-09T17:28:24.247576Z","shell.execute_reply.started":"2024-10-09T17:28:24.204670Z","shell.execute_reply":"2024-10-09T17:28:24.246635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I'll make sure that I captured all of the images in the DataFrame:","metadata":{}},{"cell_type":"code","source":"files = get_image_files(Path(\"/kaggle/working/train_spectrograms\"))\nlen(files)","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:28:24.248817Z","iopub.execute_input":"2024-10-09T17:28:24.249500Z","iopub.status.idle":"2024-10-09T17:28:24.345059Z","shell.execute_reply.started":"2024-10-09T17:28:24.249466Z","shell.execute_reply":"2024-10-09T17:28:24.344133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(unique_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:28:24.346223Z","iopub.execute_input":"2024-10-09T17:28:24.346524Z","iopub.status.idle":"2024-10-09T17:28:24.352348Z","shell.execute_reply.started":"2024-10-09T17:28:24.346500Z","shell.execute_reply":"2024-10-09T17:28:24.351390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looks good! I'll save this notebook version and then convert the output folder to a Kaggle Dataset that I can use for training.","metadata":{}},{"cell_type":"code","source":"!zip -r file.zip /kaggle/working","metadata":{"execution":{"iopub.status.busy":"2024-10-09T17:31:15.553223Z","iopub.execute_input":"2024-10-09T17:31:15.553996Z","iopub.status.idle":"2024-10-09T17:31:51.071035Z","shell.execute_reply.started":"2024-10-09T17:31:15.553964Z","shell.execute_reply":"2024-10-09T17:31:51.070030Z"},"trusted":true},"execution_count":null,"outputs":[]}]}