{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nimport os\nimport gc\nimport copy\nimport yaml\nimport random\nimport shutil\nfrom time import time\nimport typing as tp\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\n\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import StratifiedGroupKFold\n\nimport torch\nfrom torch import nn\nfrom torch import optim\nfrom torch.optim import lr_scheduler\nfrom torch.cuda import amp\n\nimport timm\n\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-21T23:04:00.656530Z","iopub.execute_input":"2024-02-21T23:04:00.657218Z","iopub.status.idle":"2024-02-21T23:04:12.324524Z","shell.execute_reply.started":"2024-02-21T23:04:00.657171Z","shell.execute_reply":"2024-02-21T23:04:12.323584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# References\n\n[🌩️HMS - EDA and Domain Journey🌍](https://www.kaggle.com/code/mvvppp/hms-eda-and-domain-journey)\n\n[HMS-HBAC: ResNet34d Baseline [Training]](https://www.kaggle.com/code/ttahara/hms-hbac-resnet34d-baseline-training)","metadata":{}},{"cell_type":"code","source":"os.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0\"","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:04:16.272460Z","iopub.execute_input":"2024-02-21T23:04:16.274195Z","iopub.status.idle":"2024-02-21T23:04:16.280457Z","shell.execute_reply.started":"2024-02-21T23:04:16.274142Z","shell.execute_reply":"2024-02-21T23:04:16.278985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = Path.cwd().parent\nINPUT = ROOT / \"input\"\nOUTPUT = ROOT / \"output\"\nSRC = ROOT / \"src\"\n\nDATA = INPUT / \"hms-harmful-brain-activity-classification\"\nTRAIN_SPEC = DATA / \"train_spectrograms\"\nTEST_SPEC = DATA / \"test_spectrograms\"\nTRAIN_EEG = DATA / \"train_eegs\"\nTEST_EEG = DATA / \"test_eegs\"\n\nTMP = ROOT / \"tmp\"\nTRAIN_SPEC_SPLIT = TMP / \"train_spectrograms_split\"\nTEST_SPEC_SPLIT = TMP / \"test_spectrograms_split\"\nTMP.mkdir(exist_ok=True)\nTRAIN_SPEC_SPLIT.mkdir(exist_ok=True)\nTEST_SPEC_SPLIT.mkdir(exist_ok=True)\n\n\nRANDAM_SEED = 1086\nCLASSES = [\"seizure_vote\", \"lpd_vote\", \"gpd_vote\", \"lrda_vote\", \"grda_vote\", \"other_vote\"]\nN_CLASSES = len(CLASSES)\nFOLDS = [0, 1, 2, 3, 4]\nN_FOLDS = len(FOLDS)","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:04:17.878595Z","iopub.execute_input":"2024-02-21T23:04:17.879306Z","iopub.status.idle":"2024-02-21T23:04:17.887725Z","shell.execute_reply.started":"2024-02-21T23:04:17.879271Z","shell.execute_reply":"2024-02-21T23:04:17.886758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(DATA / \"train.csv\")\n\n# convert vote to probability\ntrain[CLASSES] /= train[CLASSES].sum(axis=1).values[:, None]\n\nprint(train.shape)\ntrain.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:04:20.528674Z","iopub.execute_input":"2024-02-21T23:04:20.529121Z","iopub.status.idle":"2024-02-21T23:04:20.864454Z","shell.execute_reply.started":"2024-02-21T23:04:20.529087Z","shell.execute_reply":"2024-02-21T23:04:20.863071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Spectrograms","metadata":{}},{"cell_type":"markdown","source":"## Reduce Sizes\n* used the **first** `spectrogram_sub_id` for each `spectrogram_id` in order to train model faster.\n* used top 100 spectrograms for faster running","metadata":{}},{"cell_type":"code","source":"train = train.groupby(\"spectrogram_id\").head(1).reset_index(drop=True)\ntrain100 = train.head(100)\nprint(train100.shape)\ntrain100.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:04:23.879535Z","iopub.execute_input":"2024-02-21T23:04:23.880115Z","iopub.status.idle":"2024-02-21T23:04:23.924169Z","shell.execute_reply.started":"2024-02-21T23:04:23.880067Z","shell.execute_reply":"2024-02-21T23:04:23.923013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check the size of the parquet","metadata":{}},{"cell_type":"code","source":"file_paths = list(TRAIN_SPEC.glob('*.parquet'))[:20]  # Get the first 20 Parquet files\n\n# Initialize a dictionary to store the shapes and their frequencies\nshape_frequencies = {}\n\nfor file_path in file_paths:\n    df = pd.read_parquet(file_path)\n    shape = df.shape  # Get the shape of the DataFrame\n    if shape in shape_frequencies:\n        shape_frequencies[shape] += 1\n    else:\n        shape_frequencies[shape] = 1\n\n# Print the unique shapes and their frequencies\nfor shape, frequency in shape_frequencies.items():\n    print(f\"Shape: {shape}, Frequency: {frequency}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:04:25.808922Z","iopub.execute_input":"2024-02-21T23:04:25.809353Z","iopub.status.idle":"2024-02-21T23:04:27.177064Z","shell.execute_reply.started":"2024-02-21T23:04:25.809322Z","shell.execute_reply":"2024-02-21T23:04:27.175693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nshape_labels = [f\"{shape[0]}x{shape[1]}\" for shape in shape_frequencies.keys()]\nfrequencies = list(shape_frequencies.values())\npositions = range(len(shape_frequencies))\n\nplt.figure(figsize=(12, 8))  # Adjust figure size as needed\nplt.bar(positions, frequencies, align='center', alpha=0.75, color='skyblue', edgecolor='black')\n\nplt.xticks(positions, shape_labels, rotation=45)\n\nplt.title('Frequency of Unique Data Shapes')\nplt.xlabel('Data Shape (Rows x Columns)')\nplt.ylabel('Frequency')\n\nplt.tight_layout() \nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:04:29.350001Z","iopub.execute_input":"2024-02-21T23:04:29.350379Z","iopub.status.idle":"2024-02-21T23:04:29.703940Z","shell.execute_reply.started":"2024-02-21T23:04:29.350350Z","shell.execute_reply":"2024-02-21T23:04:29.702716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spec_id = '1000086677'\nparquet_file_path = (TRAIN_SPEC / f\"{spec_id}.parquet\")\ndf = pd.read_parquet(parquet_file_path)\n\n\nprint(\"Shape of the DataFrame:\", df.shape)\n\nprint(df)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:04:32.960362Z","iopub.execute_input":"2024-02-21T23:04:32.960728Z","iopub.status.idle":"2024-02-21T23:04:33.029951Z","shell.execute_reply.started":"2024-02-21T23:04:32.960699Z","shell.execute_reply":"2024-02-21T23:04:33.028709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_train_spectrogram = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000086677.parquet\")\n\ndef plot_spectrogram(spectrogram_path):\n    sample_spect = pd.read_parquet(spectrogram_path)\n    \n    split_spect = {\n        \"LL\": sample_spect.filter(regex='^LL', axis=1),\n        \"RL\": sample_spect.filter(regex='^RL', axis=1),\n        \"RP\": sample_spect.filter(regex='^RP', axis=1),\n        \"LP\": sample_spect.filter(regex='^LP', axis=1),\n    }\n    \n    fig, axes = plt.subplots(nrows=2, ncols=2, figsize=(15, 12))\n    axes = axes.flatten()\n    label_interval = 5\n    for i, split_name in enumerate(split_spect.keys()):\n        ax = axes[i]\n        img = ax.imshow(np.log(split_spect[split_name]).T, cmap='viridis', aspect='auto', origin='lower')\n        cbar = fig.colorbar(img, ax=ax)\n        cbar.set_label('Log(Value)')\n        ax.set_title(split_name)\n        ax.set_ylabel(\"Frequency (Hz)\")\n        ax.set_xlabel(\"Time\")\n\n        ax.set_yticks(np.arange(len(split_spect[split_name].columns)))\n        ax.set_yticklabels([column_name[3:] for column_name in split_spect[split_name].columns])\n        frequencies = [column_name[3:] for column_name in split_spect[split_name].columns]\n        ax.set_yticks(np.arange(0, len(split_spect[split_name].columns), label_interval))\n        ax.set_yticklabels(frequencies[::label_interval])\n    plt.tight_layout()\n    plt.show()\nplot_spectrogram(\"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/1000189855.parquet\")","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:04:35.806729Z","iopub.execute_input":"2024-02-21T23:04:35.807139Z","iopub.status.idle":"2024-02-21T23:04:38.627578Z","shell.execute_reply.started":"2024-02-21T23:04:35.807106Z","shell.execute_reply":"2024-02-21T23:04:38.626493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### In each parquet, the rows are the time stamps and columns are frequency in Hz and recording regions of the EEG electrodes. \n\n#### The latter are abbreviated as LL = left lateral; RL = right lateral; LP = left parasagittal; RP = right parasagittal. \n\n#### Time varied, but the number of recording regions of the EEG electrodes is always 400.\n\n#### The size of each parquet is not consistant (time stamp) so there's a piece of code to **extract the associated 400*300 (Hz, Time) spectrograms** associated with each train data point.","metadata":{}},{"cell_type":"markdown","source":"## Spectrograms Processing","metadata":{}},{"cell_type":"code","source":"for spec_id, df in tqdm(train100.groupby(\"spectrogram_id\")):\n    spec = pd.read_parquet(TRAIN_SPEC / f\"{spec_id}.parquet\")\n    \n    spec_arr = spec.fillna(0).values[:, 1:].T.astype(\"float32\")  # (Hz, Time) = (400, 300)\n    \n    for spec_offset, label_id in df[\n        [\"spectrogram_label_offset_seconds\", \"label_id\"]\n    ].astype(int).values:\n        spec_offset = spec_offset // 2\n        split_spec_arr = spec_arr[:, spec_offset: spec_offset + 300]\n        np.save(TRAIN_SPEC_SPLIT / f\"{label_id}.npy\" , split_spec_arr)","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:04:42.741548Z","iopub.execute_input":"2024-02-21T23:04:42.741950Z","iopub.status.idle":"2024-02-21T23:04:47.934656Z","shell.execute_reply.started":"2024-02-21T23:04:42.741918Z","shell.execute_reply":"2024-02-21T23:04:47.933597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_id = '127492639'\n\n# Assuming TRAIN_SPEC_SPLIT is a pathlib.Path or similar that specifies the directory\nfile_path = TRAIN_SPEC_SPLIT / f\"{label_id}.npy\"\n\n# Load the numpy array from the file\nloaded_spec_arr = np.load(file_path)\n\n# Get the shape of the loaded array\nshape = loaded_spec_arr.shape\n\nprint(f\"The shape of the loaded spectrogram segment is: {shape}\")\nprint(loaded_spec_arr)","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:04:49.426347Z","iopub.execute_input":"2024-02-21T23:04:49.427385Z","iopub.status.idle":"2024-02-21T23:04:49.437200Z","shell.execute_reply.started":"2024-02-21T23:04:49.427342Z","shell.execute_reply":"2024-02-21T23:04:49.435936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EGG plot","metadata":{}},{"cell_type":"code","source":"print(\"Train eeg files: \")\n! ls /kaggle/input/hms-harmful-brain-activity-classification/train_eegs | wc -l\nprint(\"Size of train eegs\")\n! du -sh /kaggle/input/hms-harmful-brain-activity-classification/train_eegs","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:05:29.525293Z","iopub.execute_input":"2024-02-21T23:05:29.526147Z","iopub.status.idle":"2024-02-21T23:05:44.387432Z","shell.execute_reply.started":"2024-02-21T23:05:29.526106Z","shell.execute_reply":"2024-02-21T23:05:44.386363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_eeg = pd.read_parquet(\"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1000913311.parquet\")\n\nprint(sample_eeg.shape)\n\nsample_eeg.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:05:46.600995Z","iopub.execute_input":"2024-02-21T23:05:46.601397Z","iopub.status.idle":"2024-02-21T23:05:46.633563Z","shell.execute_reply.started":"2024-02-21T23:05:46.601366Z","shell.execute_reply":"2024-02-21T23:05:46.632493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The column names are the names of the individual electrode locations for EEG leads.\n\nRows represent values from this electrodes over time. The frequency is 200 samples (rows) per second.\n\nSo, for this sample, this is a result of 50-second brain activity tracking.\n\nThere are always 20 electordes, but the time variaes.","metadata":{}},{"cell_type":"code","source":"file_paths = list(TRAIN_EEG.glob('*.parquet'))[:20]  # Get the first 20 Parquet files\n\n# Initialize a dictionary to store the shapes and their frequencies\nshape_frequencies = {}\n\nfor file_path in file_paths:\n    df = pd.read_parquet(file_path)\n    shape = df.shape  # Get the shape of the DataFrame\n    if shape in shape_frequencies:\n        shape_frequencies[shape] += 1\n    else:\n        shape_frequencies[shape] = 1\n        \n# Print the unique shapes and their frequencies\nfor shape, frequency in shape_frequencies.items():\n    print(f\"Shape: {shape}, Frequency: {frequency}\")","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:05:50.300166Z","iopub.execute_input":"2024-02-21T23:05:50.300575Z","iopub.status.idle":"2024-02-21T23:05:51.197867Z","shell.execute_reply.started":"2024-02-21T23:05:50.300542Z","shell.execute_reply":"2024-02-21T23:05:51.196598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nshape_labels = [f\"{shape[0]}x{shape[1]}\" for shape in shape_frequencies.keys()]\nfrequencies = list(shape_frequencies.values())\npositions = range(len(shape_frequencies))\n\nplt.figure(figsize=(12, 8))  # Adjust figure size as needed\nplt.bar(positions, frequencies, align='center', alpha=0.75, color='skyblue', edgecolor='black')\n\nplt.xticks(positions, shape_labels, rotation=45)\n\nplt.title('Frequency of Unique Data Shapes')\nplt.xlabel('Data Shape (Rows x Columns)')\nplt.ylabel('Frequency')\n\nplt.tight_layout() \nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:05:53.392922Z","iopub.execute_input":"2024-02-21T23:05:53.393319Z","iopub.status.idle":"2024-02-21T23:05:53.744158Z","shell.execute_reply.started":"2024-02-21T23:05:53.393287Z","shell.execute_reply":"2024-02-21T23:05:53.742789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(20, figsize=(10, 100))\n\n# Generate a line plot for each column in the DataFrame\nfor i, column in enumerate(sample_eeg.columns):\n    ax[i].plot(sample_eeg.index, sample_eeg[column], label=column)\n    ax[i].grid(True)\n    ax[i].set_title(str(column))\n\n# plt.legend()\n# plt.title('Simulated Data Line Chart')\n# plt.xlabel('Index')\n# plt.ylabel('Values')\n# plt.grid(True)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-02-21T23:06:08.807354Z","iopub.execute_input":"2024-02-21T23:06:08.810311Z","iopub.status.idle":"2024-02-21T23:06:16.692317Z","shell.execute_reply.started":"2024-02-21T23:06:08.810267Z","shell.execute_reply":"2024-02-21T23:06:16.691060Z"},"trusted":true},"execution_count":null,"outputs":[]}]}