{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-14T07:20:23.964203Z","iopub.execute_input":"2024-03-14T07:20:23.964752Z","iopub.status.idle":"2024-03-14T07:20:33.87574Z","shell.execute_reply.started":"2024-03-14T07:20:23.964709Z","shell.execute_reply":"2024-03-14T07:20:33.874303Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install pyriemann\n","metadata":{"execution":{"iopub.status.busy":"2024-03-15T18:37:50.580954Z","iopub.execute_input":"2024-03-15T18:37:50.581954Z","iopub.status.idle":"2024-03-15T18:38:07.68643Z","shell.execute_reply.started":"2024-03-15T18:37:50.581913Z","shell.execute_reply":"2024-03-15T18:38:07.685182Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\nimport sys\nimport math\nimport time\nimport datetime as dt\n\nfrom glob import glob\nfrom pathlib import Path\nfrom typing import Dict, List, Union\nimport scipy.signal as scisig\nfrom scipy.signal import butter, lfilter, freqz\nfrom matplotlib import pyplot as plt\nfrom tqdm.auto import tqdm\nimport joblib","metadata":{"execution":{"iopub.status.busy":"2024-03-14T07:20:33.878373Z","iopub.execute_input":"2024-03-14T07:20:33.878731Z","iopub.status.idle":"2024-03-14T07:20:33.885281Z","shell.execute_reply.started":"2024-03-14T07:20:33.878702Z","shell.execute_reply":"2024-03-14T07:20:33.883966Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class APP:\n    jupyter = \"ipykernel\" in globals()\n    if not jupyter:\n        try:\n            if \"IPython\" in globals().get(\"__doc__\", \"\"):\n                jupyter = True\n        except Exception as inst:\n            print(inst)\n\n    kaggle = os.environ.get(\"KAGGLE_KERNEL_RUN_TYPE\", \"\") != \"\"\n    local = os.environ.get(\"DOCKER_USING\", \"\") == \"LOCAL\"\n    date_time_start = dt.datetime.now()\n    dt_start_ymd_hms = date_time_start.strftime(\"%Y.%m.%d_%H-%M-%S\")\n\n    file_run_path = \"\"\n    if jupyter:\n        try:\n            file_run_path = Path(globals().get(\"__vsc_ipynb_file__\", \"\"))\n        except Exception as inst:\n            print(inst)\n\n    else:\n        try:\n            file_run_path = Path(__file__)\n        except Exception as inst:\n            print(inst)\n\n    file_run_name = file_run_path.stem\n    path_app = file_run_path.parent\n    path_run = Path(os.getcwd())\n    path_out = (\n        Path(\"/kaggle/working\")\n        if kaggle\n        else file_run_path / f\"{file_run_name}_{dt_start_ymd_hms}\"\n    )\n\n\nOUTPUT_DIR = \"./\"\nif not os.path.exists(OUTPUT_DIR):\n    os.makedirs(OUTPUT_DIR)\n\nprint(f\"jupyter:{APP.jupyter}, kaggle:{APP.kaggle}, local:{APP.local}\")\nprint(APP.file_run_path)\nprint(APP.path_out)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T07:20:33.886763Z","iopub.execute_input":"2024-03-14T07:20:33.887597Z","iopub.status.idle":"2024-03-14T07:20:33.905475Z","shell.execute_reply.started":"2024-03-14T07:20:33.887564Z","shell.execute_reply":"2024-03-14T07:20:33.904547Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    verbose = 1  # Verbosity\n    seed = 42  # Random seed\n    preset = \"efficientnetv2_b2_imagenet\"  # Name of pretrained classifier\n    image_size = [400, 300]  # Input image size\n    epochs = 13 # Training epochs\n    batch_size = 64  # Batch size\n    lr_mode = \"cos\" # LR scheduler mode from one of \"cos\", \"step\", \"exp\"\n    drop_remainder = True  # Drop incomplete batches\n    num_classes = 6 # Number of classes in the dataset\n    fold = 0 # Which fold to set as validation data\n    class_names = ['Seizure', 'LPD', 'GPD', 'LRDA','GRDA', 'Other']\n    label2name = dict(enumerate(class_names))\n    name2label = {v:k for k, v in label2name.items()}","metadata":{"execution":{"iopub.status.busy":"2024-03-14T07:20:33.908187Z","iopub.execute_input":"2024-03-14T07:20:33.908551Z","iopub.status.idle":"2024-03-14T07:20:33.917392Z","shell.execute_reply.started":"2024-03-14T07:20:33.90852Z","shell.execute_reply":"2024-03-14T07:20:33.916281Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification\"\n\nSPEC_DIR = \"/tmp/dataset/hms-hbac\"\nos.makedirs(SPEC_DIR+'/train_spectrograms', exist_ok=True)\nos.makedirs(SPEC_DIR+'/test_spectrograms', exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T07:20:33.918717Z","iopub.execute_input":"2024-03-14T07:20:33.919072Z","iopub.status.idle":"2024-03-14T07:20:33.930797Z","shell.execute_reply.started":"2024-03-14T07:20:33.919042Z","shell.execute_reply":"2024-03-14T07:20:33.929704Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(f'{BASE_PATH}/train.csv')\ndf['eeg_path'] = f'{BASE_PATH}/train_eegs/'+df['eeg_id'].astype(str)+'.parquet'\ndf['spec_path'] = f'{BASE_PATH}/train_spectrograms/'+df['spectrogram_id'].astype(str)+'.parquet'\ndf['spec2_path'] = f'{SPEC_DIR}/train_spectrograms/'+df['spectrogram_id'].astype(str)+'.npy'\ndf['class_name'] = df.expert_consensus.copy()\ndf['class_label'] = df.expert_consensus.map(CFG.name2label)\ndisplay(df.head(5))","metadata":{"execution":{"iopub.status.busy":"2024-03-14T07:20:33.93221Z","iopub.execute_input":"2024-03-14T07:20:33.932859Z","iopub.status.idle":"2024-03-14T07:20:34.600536Z","shell.execute_reply.started":"2024-03-14T07:20:33.932828Z","shell.execute_reply":"2024-03-14T07:20:34.599357Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Test\ntest_df = pd.read_csv(f'{BASE_PATH}/test.csv')\ntest_df['eeg_path'] = f'{BASE_PATH}/test_eegs/'+test_df['eeg_id'].astype(str)+'.parquet'\ntest_df['spec_path'] = f'{BASE_PATH}/test_spectrograms/'+test_df['spectrogram_id'].astype(str)+'.parquet'\ntest_df['spec2_path'] = f'{SPEC_DIR}/test_spectrograms/'+test_df['spectrogram_id'].astype(str)+'.npy'\ndisplay(test_df.head(2))","metadata":{"execution":{"iopub.status.busy":"2024-03-14T07:20:34.602194Z","iopub.execute_input":"2024-03-14T07:20:34.602844Z","iopub.status.idle":"2024-03-14T07:20:34.624413Z","shell.execute_reply.started":"2024-03-14T07:20:34.602799Z","shell.execute_reply":"2024-03-14T07:20:34.623321Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Conversion from .parquet to .npy","metadata":{}},{"cell_type":"code","source":"# Define a function to process a single eeg_id\ndef process_spec(spec_id, split=\"train\"):\n    spec_path = f\"{BASE_PATH}/{split}_spectrograms/{spec_id}.parquet\"\n    spec = pd.read_parquet(spec_path)\n    spec = spec.fillna(0).values[:, 1:].T  # fill NaN values with 0, transpose for (Time, Freq) -> (Freq, Time)\n    spec = spec.astype(\"float32\")\n    np.save(f\"{SPEC_DIR}/{split}_spectrograms/{spec_id}.npy\", spec)\n\n# Get unique spec_ids of train and valid data\nspec_ids = df[\"spectrogram_id\"].unique()\n\n# Parallelize the processing using joblib for training data with the threading backend\n_ = joblib.Parallel(n_jobs=-1, backend=\"threading\")(\n    joblib.delayed(process_spec)(spec_id, \"train\")\n    for spec_id in tqdm(spec_ids, total=len(spec_ids))\n)\n\n# Get unique spec_ids of test data\ntest_spec_ids = test_df[\"spectrogram_id\"].unique()\n\n# Parallelize the processing using joblib for test data with the threading backend\n_ = joblib.Parallel(n_jobs=-1, backend=\"threading\")(\n    joblib.delayed(process_spec)(spec_id, \"test\")\n    for spec_id in tqdm(test_spec_ids, total=len(test_spec_ids))\n)","metadata":{"execution":{"iopub.status.busy":"2024-03-14T07:33:25.32767Z","iopub.execute_input":"2024-03-14T07:33:25.328235Z","iopub.status.idle":"2024-03-14T07:38:26.492863Z","shell.execute_reply.started":"2024-03-14T07:33:25.328198Z","shell.execute_reply":"2024-03-14T07:38:26.491882Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport tensorflow as tf\nfrom sklearn.preprocessing import StandardScaler\n\n# Define your CFG class and other necessary imports\n\n# ... (Define CFG class and other imports)\n\ndef build_decoder_scikit_learn(with_labels=True, target_size=CFG.image_size, dtype=32):\n    def decode_signal(path, offset=None):\n        # Read .npy files and process the signal\n        file_bytes = tf.io.read_file(path)\n        sig = tf.io.decode_raw(file_bytes, tf.float32)\n        sig = sig[1024//dtype:]  # Remove header tag\n        sig = tf.reshape(sig, [400, -1])\n\n        # Extract labeled subsample from the full spectrogram using \"offset\"\n        if offset is not None: \n            offset = offset // 2  # Only odd values are given\n            sig = sig[:, offset:offset+300]\n            \n            # Pad spectrogram to ensure the same input shape of [400, 300]\n            pad_size = tf.math.maximum(0, 300 - tf.shape(sig)[1])\n            sig = tf.pad(sig, [[0, 0], [0, pad_size]])\n            sig = tf.reshape(sig, [400, 300])\n        \n        # Log spectrogram \n        sig = tf.clip_by_value(sig, tf.math.exp(-4.0), tf.math.exp(8.0)) # avoid 0 in log\n        sig = tf.math.log(sig)\n\n        # Normalize spectrogram using StandardScaler\n        scaler = StandardScaler()\n        sig = scaler.fit_transform(sig.numpy().T).T\n\n        # Mono channel to 3 channels to use \"ImageNet\" weights\n        sig = tf.tile(sig[..., None], [1, 1, 3])\n        return sig\n\n    # ... (remaining code remains the same)\n    def decode_label(label):\n        label = tf.one_hot(label, CFG.num_classes)\n        label = tf.cast(label, tf.float32)\n        label = tf.reshape(label, [CFG.num_classes])\n        return label\n    \n    def decode_with_labels(path, offset=None, label=None):\n        sig = decode_signal(path, offset)\n        label = decode_label(label)\n        return (sig, label)\n    \n    return decode_with_labels if with_labels else decode_signal\n\n# ... (Define build_augmenter and other functions)\n\ndef build_dataset(paths, offsets=None, labels=None, batch_size=32, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=False, repeat=True, shuffle=1024, \n                  cache_dir=\"\", drop_remainder=False):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    \n    if decode_fn is None:\n        decode_fn = build_decoder_scikit_learn(labels is not None)\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = (paths, offsets) if labels is None else (paths, offsets, labels)\n    \n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    ds = ds.map(decode_fn, num_parallel_calls=AUTO)\n    ds = ds.cache(cache_dir) if cache else ds\n    ds = ds.repeat() if repeat else ds\n    if shuffle: \n        ds = ds.shuffle(shuffle, seed=CFG.seed)\n        opt = tf.data.Options()\n        opt.experimental_deterministic = False\n        ds = ds.with_options(opt)\n    ds = ds.batch(batch_size, drop_remainder=drop_remainder)\n    ds = ds.map(augment_fn, num_parallel_calls=AUTO) if augment else ds\n    ds = ds.prefetch(AUTO)\n    return ds\n","metadata":{"execution":{"iopub.status.busy":"2024-03-14T08:21:33.368778Z","iopub.execute_input":"2024-03-14T08:21:33.369282Z","iopub.status.idle":"2024-03-14T08:21:33.394589Z","shell.execute_reply.started":"2024-03-14T08:21:33.369245Z","shell.execute_reply":"2024-03-14T08:21:33.39336Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedGroupKFold\n\nsgkf = StratifiedGroupKFold(n_splits=5, shuffle=True, random_state=CFG.seed)\n\ndf[\"fold\"] = -1\ndf.reset_index(drop=True, inplace=True)\nfor fold, (train_idx, valid_idx) in enumerate(\n    sgkf.split(df, y=df[\"class_label\"], groups=df[\"patient_id\"])\n):\n    df.loc[valid_idx, \"fold\"] = fold\ndf.groupby([\"fold\", \"class_name\"])[[\"eeg_id\"]].count().T","metadata":{"execution":{"iopub.status.busy":"2024-03-14T08:21:35.723068Z","iopub.execute_input":"2024-03-14T08:21:35.724243Z","iopub.status.idle":"2024-03-14T08:21:37.076285Z","shell.execute_reply.started":"2024-03-14T08:21:35.724188Z","shell.execute_reply":"2024-03-14T08:21:37.075131Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sample from full data\nsample_df = df.groupby(\"spectrogram_id\").head(1).reset_index(drop=True)\ntrain_df = sample_df[sample_df.fold != CFG.fold]\nvalid_df = sample_df[sample_df.fold == CFG.fold]\nprint(f\"# Num Train: {len(train_df)} | Num Valid: {len(valid_df)}\")\n\n# Train\ntrain_paths = train_df.spec2_path.values\ntrain_offsets = train_df.spectrogram_label_offset_seconds.values.astype(int)\ntrain_labels = train_df.class_label.values\ntrain_ds = build_dataset(train_paths,train_offsets,train_labels,batch_size=CFG.batch_size,repeat=True,shuffle=True, cache=True)\n\n# Valid\nvalid_paths = valid_df.spec2_path.values\nvalid_offsets = valid_df.spectrogram_label_offset_seconds.values.astype(int)\nvalid_labels = valid_df.class_label.values\nvalid_ds = build_dataset(valid_paths,valid_offsets,valid_labels,batch_size=CFG.batch_size,repeat=False,shuffle=False, cache=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-14T08:21:39.879203Z","iopub.execute_input":"2024-03-14T08:21:39.879588Z","iopub.status.idle":"2024-03-14T08:21:40.533361Z","shell.execute_reply.started":"2024-03-14T08:21:39.879559Z","shell.execute_reply":"2024-03-14T08:21:40.531455Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"imgs, tars = next(iter(train_ds))\n\nnum_imgs = 8\nplt.figure(figsize=(4*4, num_imgs//4*5))\nfor i in range(num_imgs):\n    plt.subplot(num_imgs//4, 4, i + 1)\n    img = imgs[i].numpy()[...,0]  # Adjust as per your image data format\n    img -= img.min()\n    img /= img.max() + 1e-4\n    tar = CFG.label2name[np.argmax(tars[i].numpy())]\n    plt.imshow(img)\n    plt.title(f\"Target: {tar}\")\n    plt.axis('off')\n    \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-14T07:39:08.29883Z","iopub.execute_input":"2024-03-14T07:39:08.299603Z","iopub.status.idle":"2024-03-14T07:39:08.357716Z","shell.execute_reply.started":"2024-03-14T07:39:08.299565Z","shell.execute_reply":"2024-03-14T07:39:08.355962Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}