{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":59093,"databundleVersionId":7469972},{"sourceType":"modelInstanceVersion","sourceId":6127,"databundleVersionId":7429415,"modelInstanceId":4598}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# HMS - Harmful Brain Activity Classification using PySpark (Mixed Pipeline)\n\n# ----------------------------------------------\n# Install Libraries - Not needed for PySpark core in Kaggle\n\n# ----------------------------------------------\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"spark","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T19:32:39.015237Z","iopub.execute_input":"2025-06-25T19:32:39.015739Z","iopub.status.idle":"2025-06-25T19:32:39.062242Z","shell.execute_reply.started":"2025-06-25T19:32:39.01563Z","shell.execute_reply":"2025-06-25T19:32:39.060809Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Import Libraries\nfrom pyspark.sql import SparkSession\nfrom pyspark.sql.functions import col, concat, lit\nimport pandas as pd\nimport numpy as np\nimport joblib\nimport os\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:10:17.830283Z","iopub.execute_input":"2025-06-25T17:10:17.830572Z","iopub.status.idle":"2025-06-25T17:10:21.206741Z","shell.execute_reply.started":"2025-06-25T17:10:17.830549Z","shell.execute_reply":"2025-06-25T17:10:21.205578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Configuration\nclass CFG:\n    seed = 42\n    image_size = [400, 300]\n    batch_size = 64\n    epochs = 13\n    num_classes = 6\n    fold = 0\n    class_names = ['Seizure', 'LPD', 'GPD', 'LRDA', 'GRDA', 'Other']\n    label2name = dict(enumerate(class_names))\n    name2label = {v: k for k, v in label2name.items()}\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:10:24.788671Z","iopub.execute_input":"2025-06-25T17:10:24.789053Z","iopub.status.idle":"2025-06-25T17:10:24.795161Z","shell.execute_reply.started":"2025-06-25T17:10:24.789017Z","shell.execute_reply":"2025-06-25T17:10:24.794099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Reproducibility\nimport random\nrandom.seed(CFG.seed)\nnp.random.seed(CFG.seed)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:10:34.975233Z","iopub.execute_input":"2025-06-25T17:10:34.976382Z","iopub.status.idle":"2025-06-25T17:10:34.98096Z","shell.execute_reply.started":"2025-06-25T17:10:34.976347Z","shell.execute_reply":"2025-06-25T17:10:34.979891Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dataset Path\nBASE_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification\"\nSPEC_DIR = \"/tmp/dataset/hms-hbac\"\nos.makedirs(SPEC_DIR+'/train_spectrograms', exist_ok=True)\nos.makedirs(SPEC_DIR+'/test_spectrograms', exist_ok=True)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:12:04.693443Z","iopub.execute_input":"2025-06-25T17:12:04.694576Z","iopub.status.idle":"2025-06-25T17:12:04.700431Z","shell.execute_reply.started":"2025-06-25T17:12:04.694536Z","shell.execute_reply":"2025-06-25T17:12:04.69921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Start PySpark Session\nspark = SparkSession.builder.appName(\"HMS EEG Spark\").getOrCreate()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:12:07.95654Z","iopub.execute_input":"2025-06-25T17:12:07.956881Z","iopub.status.idle":"2025-06-25T17:12:07.966805Z","shell.execute_reply.started":"2025-06-25T17:12:07.956858Z","shell.execute_reply":"2025-06-25T17:12:07.965486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Meta Data Loading\ntrain_df = spark.read.csv(f\"{BASE_PATH}/train.csv\", header=True, inferSchema=True)\ntest_df = spark.read.csv(f\"{BASE_PATH}/test.csv\", header=True, inferSchema=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:12:23.033896Z","iopub.execute_input":"2025-06-25T17:12:23.034197Z","iopub.status.idle":"2025-06-25T17:12:24.718895Z","shell.execute_reply.started":"2025-06-25T17:12:23.034177Z","shell.execute_reply":"2025-06-25T17:12:24.717989Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.printSchema()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:12:41.218711Z","iopub.execute_input":"2025-06-25T17:12:41.219082Z","iopub.status.idle":"2025-06-25T17:12:41.226573Z","shell.execute_reply.started":"2025-06-25T17:12:41.219058Z","shell.execute_reply":"2025-06-25T17:12:41.225604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.printSchema()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:12:52.030523Z","iopub.execute_input":"2025-06-25T17:12:52.030872Z","iopub.status.idle":"2025-06-25T17:12:52.038137Z","shell.execute_reply.started":"2025-06-25T17:12:52.03085Z","shell.execute_reply":"2025-06-25T17:12:52.036747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.show(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:13:11.851561Z","iopub.execute_input":"2025-06-25T17:13:11.851899Z","iopub.status.idle":"2025-06-25T17:13:12.064528Z","shell.execute_reply.started":"2025-06-25T17:13:11.851876Z","shell.execute_reply":"2025-06-25T17:13:12.063646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:13:20.841364Z","iopub.execute_input":"2025-06-25T17:13:20.841694Z","iopub.status.idle":"2025-06-25T17:13:21.104412Z","shell.execute_reply.started":"2025-06-25T17:13:20.841671Z","shell.execute_reply":"2025-06-25T17:13:21.103505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Add paths\ntrain_df = train_df \\\n    .withColumn(\"eeg_path\", concat(lit(BASE_PATH + \"/train_eegs/\"), col(\"eeg_id\"), lit(\".parquet\"))) \\\n    .withColumn(\"spec_path\", concat(lit(BASE_PATH + \"/train_spectrograms/\"), col(\"spectrogram_id\"), lit(\".parquet\"))) \\\n    .withColumn(\"spec2_path\", concat(lit(SPEC_DIR + \"/train_spectrograms/\"), col(\"spectrogram_id\"), lit(\".npy\")))\n\ntest_df = test_df \\\n    .withColumn(\"eeg_path\", concat(lit(BASE_PATH + \"/test_eegs/\"), col(\"eeg_id\"), lit(\".parquet\"))) \\\n    .withColumn(\"spec_path\", concat(lit(BASE_PATH + \"/test_spectrograms/\"), col(\"spectrogram_id\"), lit(\".parquet\"))) \\\n    .withColumn(\"spec2_path\", concat(lit(SPEC_DIR + \"/test_spectrograms/\"), col(\"spectrogram_id\"), lit(\".npy\")))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:13:25.96058Z","iopub.execute_input":"2025-06-25T17:13:25.961504Z","iopub.status.idle":"2025-06-25T17:13:26.172977Z","shell.execute_reply.started":"2025-06-25T17:13:25.961472Z","shell.execute_reply":"2025-06-25T17:13:26.171906Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# New columns(eeg_path, spec_path, spec2_path) are added in trian_df\ntrain_df.printSchema()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:14:13.058139Z","iopub.execute_input":"2025-06-25T17:14:13.058486Z","iopub.status.idle":"2025-06-25T17:14:13.065158Z","shell.execute_reply.started":"2025-06-25T17:14:13.058461Z","shell.execute_reply":"2025-06-25T17:14:13.064064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.show(3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:16:35.159446Z","iopub.execute_input":"2025-06-25T17:16:35.159813Z","iopub.status.idle":"2025-06-25T17:16:35.460277Z","shell.execute_reply.started":"2025-06-25T17:16:35.159765Z","shell.execute_reply":"2025-06-25T17:16:35.458969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# New columns(eeg_path, spec_path, spec2_path) are added in test_df\ntest_df.printSchema()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:16:02.069768Z","iopub.execute_input":"2025-06-25T17:16:02.070261Z","iopub.status.idle":"2025-06-25T17:16:02.077763Z","shell.execute_reply.started":"2025-06-25T17:16:02.070233Z","shell.execute_reply":"2025-06-25T17:16:02.076415Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:16:16.825744Z","iopub.execute_input":"2025-06-25T17:16:16.826197Z","iopub.status.idle":"2025-06-25T17:16:17.154902Z","shell.execute_reply.started":"2025-06-25T17:16:16.826158Z","shell.execute_reply":"2025-06-25T17:16:17.153287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert to Pandas\ntrain_pd = train_df.toPandas()\ntest_pd = test_df.toPandas()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:17:01.271725Z","iopub.execute_input":"2025-06-25T17:17:01.272128Z","iopub.status.idle":"2025-06-25T17:17:04.562111Z","shell.execute_reply.started":"2025-06-25T17:17:01.272092Z","shell.execute_reply":"2025-06-25T17:17:04.56109Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_pd['class_name'] = train_pd['expert_consensus']\ntrain_pd['class_label'] = train_pd['class_name'].map(CFG.name2label)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:17:10.424418Z","iopub.execute_input":"2025-06-25T17:17:10.424735Z","iopub.status.idle":"2025-06-25T17:17:10.453645Z","shell.execute_reply.started":"2025-06-25T17:17:10.424713Z","shell.execute_reply":"2025-06-25T17:17:10.452622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Convert .parquet to .npy( can store wide range of data types) for Train & Test\n\ndef process_spec(spec_id, split=\"train\"):\n    path = f\"{BASE_PATH}/{split}_spectrograms/{spec_id}.parquet\"\n    spec = pd.read_parquet(path).fillna(0).values[:, 1:].T.astype(\"float32\")\n    np.save(f\"{SPEC_DIR}/{split}_spectrograms/{spec_id}.npy\", spec)\n\n_ = joblib.Parallel(n_jobs=-1)(joblib.delayed(process_spec)(sid, \"train\") for sid in tqdm(train_pd[\"spectrogram_id\"].unique()))\n_ = joblib.Parallel(n_jobs=-1)(joblib.delayed(process_spec)(sid, \"test\") for sid in tqdm(test_pd[\"spectrogram_id\"].unique()))\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:18:50.327502Z","iopub.execute_input":"2025-06-25T17:18:50.327837Z","iopub.status.idle":"2025-06-25T17:22:36.435577Z","shell.execute_reply.started":"2025-06-25T17:18:50.327806Z","shell.execute_reply":"2025-06-25T17:22:36.434215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualization (PySpark can't plot directly)\n# Convert label count to Pandas for plotting\nlabel_dist = train_pd['class_name'].value_counts()\nlabel_dist.plot(kind='bar', title='Class Distribution', ylabel='Samples', xlabel='Class')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:22:44.110616Z","iopub.execute_input":"2025-06-25T17:22:44.110971Z","iopub.status.idle":"2025-06-25T17:22:44.68858Z","shell.execute_reply.started":"2025-06-25T17:22:44.110945Z","shell.execute_reply":"2025-06-25T17:22:44.686452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Visualize Class Distribution (Bar Chart)\n# plt.figure(figsize=(8, 5))\n# train_pd['class_name'].value_counts().plot(kind='bar', color='skyblue')\n# plt.title(\"Class Distribution\")\n# plt.xlabel(\"Brain Activity Class\")\n# plt.ylabel(\"Count\")\n# plt.xticks(rotation=45)\n# plt.grid(axis='y')\n# plt.tight_layout()\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:23:59.174856Z","iopub.execute_input":"2025-06-25T17:23:59.175258Z","iopub.status.idle":"2025-06-25T17:23:59.180331Z","shell.execute_reply.started":"2025-06-25T17:23:59.175231Z","shell.execute_reply":"2025-06-25T17:23:59.179344Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ----------------------------------------------\n# Further steps like DataLoader, Modeling, Training, Submission are done in Keras/TensorFlow\n# Continue training in TF using preprocessed npy files\n\n# Done: PySpark parts of the pipeline (Loading, Processing, Metadata, Saving, Visualization)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Since we’ve completed the PySpark steps (like loading, processing, saving `.npy`, and metadata), \n# let’s now continue the rest of pipeline using TensorFlow/Keras — as modeling and training can't be done in PySpark.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T19:30:39.465964Z","iopub.status.idle":"2025-06-25T19:30:39.466485Z","shell.execute_reply.started":"2025-06-25T19:30:39.466173Z","shell.execute_reply":"2025-06-25T19:30:39.46622Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport numpy as np\nimport keras_cv\nimport keras\nimport os","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:41:56.564522Z","iopub.execute_input":"2025-06-25T17:41:56.564895Z","iopub.status.idle":"2025-06-25T17:41:56.569981Z","shell.execute_reply.started":"2025-06-25T17:41:56.564871Z","shell.execute_reply":"2025-06-25T17:41:56.568647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Configuration and Reproducibility\n\nclass CFG:\n    verbose = 1\n    seed = 42\n    preset = \"efficientnetv2_b2_imagenet\"\n    image_size = [400, 300]\n    epochs = 13\n    batch_size = 64\n    lr_mode = \"cos\"\n    drop_remainder = True\n    num_classes = 6\n    fold = 0\n    class_names = ['Seizure', 'LPD', 'GPD', 'LRDA','GRDA', 'Other']\n    label2name = dict(enumerate(class_names))\n    name2label = {v:k for k, v in label2name.items()}\n    \n\nkeras.utils.set_random_seed(CFG.seed)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:42:27.523618Z","iopub.execute_input":"2025-06-25T17:42:27.523968Z","iopub.status.idle":"2025-06-25T17:42:27.531513Z","shell.execute_reply.started":"2025-06-25T17:42:27.523942Z","shell.execute_reply":"2025-06-25T17:42:27.530534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Dataset Split (StratifiedGroupKFold)\n\nfrom sklearn.model_selection import StratifiedGroupKFold\n\ndf = train_pd.copy()\nsgkf = StratifiedGroupKFold(n_splits=5, shuffle=True, random_state=CFG.seed)\ndf[\"fold\"] = -1\ndf.reset_index(drop=True, inplace=True)\n\nfor fold, (train_idx, valid_idx) in enumerate(\n    sgkf.split(df, y=df[\"class_label\"], groups=df[\"patient_id\"])\n):\n    df.loc[valid_idx, \"fold\"] = fold","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:34:33.666517Z","iopub.execute_input":"2025-06-25T17:34:33.666902Z","iopub.status.idle":"2025-06-25T17:34:34.785457Z","shell.execute_reply.started":"2025-06-25T17:34:33.666874Z","shell.execute_reply":"2025-06-25T17:34:34.784283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Build TensorFlow Dataset\n\ndef decode_signal(path, offset=0):\n    sig = np.load(path.decode()).astype(\"float32\")\n    sig = sig[:, offset*10:offset*10+300]\n    return sig\n\ndef build_augmenter():\n    def augment(sig, label):\n        sig = tf.clip_by_value(sig, tf.math.exp(-4.0), tf.math.exp(8.0))\n        sig = tf.math.log(sig)\n        sig -= tf.math.reduce_mean(sig)\n        sig /= tf.math.reduce_std(sig) + 1e-6\n        sig = tf.tile(sig[..., None], [1, 1, 3])\n        return sig, label\n    return augment\n\ndef build_dataset(paths, offsets, labels, batch_size, augment=False, repeat=True, shuffle=True):\n    path_ds = tf.data.Dataset.from_tensor_slices(paths)\n    offset_ds = tf.data.Dataset.from_tensor_slices(offsets)\n    label_ds = tf.data.Dataset.from_tensor_slices(tf.one_hot(labels, CFG.num_classes))\n\n    def process(path, offset, label):\n        sig = tf.numpy_function(decode_signal, [path, offset], tf.float32)\n        sig.set_shape([400, 300])\n        return sig, label\n\n    ds = tf.data.Dataset.zip((path_ds, offset_ds, label_ds))\n    ds = ds.map(process)\n    if repeat: ds = ds.repeat()\n    if shuffle: ds = ds.shuffle(2048, seed=CFG.seed)\n    if augment: ds = ds.map(build_augmenter())\n    ds = ds.batch(batch_size)\n    ds = ds.prefetch(tf.data.AUTOTUNE)\n    return ds","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:42:33.89809Z","iopub.execute_input":"2025-06-25T17:42:33.898467Z","iopub.status.idle":"2025-06-25T17:42:33.910004Z","shell.execute_reply.started":"2025-06-25T17:42:33.89844Z","shell.execute_reply":"2025-06-25T17:42:33.908882Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare Train & Valid Dataset\n\nsample_df = df.groupby(\"spectrogram_id\").head(1).reset_index(drop=True)\ntrain_df_f = sample_df[sample_df.fold != CFG.fold]\nvalid_df_f = sample_df[sample_df.fold == CFG.fold]\n\ntrain_ds = build_dataset(\n    train_df_f.spec2_path.values,\n    train_df_f.spectrogram_label_offset_seconds.values.astype(int),\n    train_df_f.class_label.values,\n    CFG.batch_size,\n    augment=True\n)\n\nvalid_ds = build_dataset(\n    valid_df_f.spec2_path.values,\n    valid_df_f.spectrogram_label_offset_seconds.values.astype(int),\n    valid_df_f.class_label.values,\n    CFG.batch_size,\n    repeat=False,\n    shuffle=False\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:42:36.364132Z","iopub.execute_input":"2025-06-25T17:42:36.364496Z","iopub.status.idle":"2025-06-25T17:42:36.525661Z","shell.execute_reply.started":"2025-06-25T17:42:36.364473Z","shell.execute_reply":"2025-06-25T17:42:36.524488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Modeling\n\nmodel = keras_cv.models.ImageClassifier.from_preset(\n    CFG.preset,\n    num_classes=CFG.num_classes\n)\n\nmodel.compile(\n    optimizer=keras.optimizers.Adam(learning_rate=1e-4),\n    loss=keras.losses.KLDivergence()\n)\n\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:42:39.177919Z","iopub.execute_input":"2025-06-25T17:42:39.178253Z","iopub.status.idle":"2025-06-25T17:43:01.462508Z","shell.execute_reply.started":"2025-06-25T17:42:39.178233Z","shell.execute_reply":"2025-06-25T17:43:01.461387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Callbacks: LR Scheduler & Checkpointing\n\nimport math\n\ndef get_lr_callback(batch_size=64, mode='cos', epochs=13):\n    lr_start, lr_max, lr_min = 5e-5, 6e-6 * batch_size, 1e-5\n    lr_ramp_ep, lr_sus_ep, lr_decay = 3, 0, 0.75\n\n    def lrfn(epoch):\n        if epoch < lr_ramp_ep: return (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n        elif mode == 'cos':\n            decay_total_epochs = epochs - lr_ramp_ep + 3\n            phase = math.pi * (epoch - lr_ramp_ep) / decay_total_epochs\n            return (lr_max - lr_min) * 0.5 * (1 + math.cos(phase)) + lr_min\n        else:\n            return lr_max\n\n    return keras.callbacks.LearningRateScheduler(lrfn)\n\nlr_cb = get_lr_callback(CFG.batch_size, mode=CFG.lr_mode, epochs=CFG.epochs)\nckpt_cb = keras.callbacks.ModelCheckpoint(\"best_model.keras\", monitor='val_loss', save_best_only=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T17:43:45.881948Z","iopub.execute_input":"2025-06-25T17:43:45.882304Z","iopub.status.idle":"2025-06-25T17:43:45.89105Z","shell.execute_reply.started":"2025-06-25T17:43:45.882281Z","shell.execute_reply":"2025-06-25T17:43:45.890093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\n\n# Function to read .npy file and slice based on offset\ndef read_npy_file_with_offset(path, offset):\n    import numpy as np\n    path = path.decode()  # convert bytes to string\n    data = np.load(path)  # shape: (freq, time) = (400, ≥300)\n    return data[:, offset:offset + 300]  # final shape (400, 300)\n\n# Decode function for TensorFlow pipeline\ndef decode_signal(path, offset):\n    # Load raw npy slice\n    sig = tf.numpy_function(read_npy_file_with_offset, [path, offset], tf.float32)\n    sig.set_shape([400, 300])  # set shape explicitly\n\n    # ✅ Expand to (400, 300, 3)\n    sig = tf.expand_dims(sig, axis=-1)  # (400, 300, 1)\n    sig = tf.tile(sig, [1, 1, 3])       # (400, 300, 3)\n\n    return sig\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T19:36:06.233089Z","iopub.execute_input":"2025-06-25T19:36:06.233479Z","iopub.status.idle":"2025-06-25T19:36:06.240908Z","shell.execute_reply.started":"2025-06-25T19:36:06.233454Z","shell.execute_reply":"2025-06-25T19:36:06.239502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train Model\n\nhistory = model.fit(\n    train_ds,\n    validation_data=valid_ds,\n    epochs=CFG.epochs,\n    steps_per_epoch=len(train_df_f)//CFG.batch_size,\n    callbacks=[lr_cb, ckpt_cb],\n    verbose=CFG.verbose\n)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T19:36:16.685726Z","iopub.execute_input":"2025-06-25T19:36:16.686684Z","iopub.status.idle":"2025-06-25T19:37:02.07304Z","shell.execute_reply.started":"2025-06-25T19:36:16.686649Z","shell.execute_reply":"2025-06-25T19:37:02.071184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prediction and Submission\n\nmodel.load_weights(\"best_model.keras\")\n\n# Build test dataset\ntest_paths = test_df_pd[\"spec2_path\"].values\ntest_ds = build_dataset(test_paths, np.zeros(len(test_paths)), np.zeros(len(test_paths)),\n                        batch_size=CFG.batch_size, repeat=False, shuffle=False)\n\n# Predict\npreds = model.predict(test_ds)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-25T19:37:02.073989Z","iopub.status.idle":"2025-06-25T19:37:02.074358Z","shell.execute_reply.started":"2025-06-25T19:37:02.074193Z","shell.execute_reply":"2025-06-25T19:37:02.074207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Create submission\nsub_df = pd.read_csv(f\"{BASE_PATH}/sample_submission.csv\")\nsub_df.iloc[:, 1:] = preds\nsub_df.to_csv(\"submission.csv\", index=False)\nsub_df.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load best model\nmodel.load_weights(\"best_model.keras\")\n\n# Build test dataset\ntest_ds = build_dataset(test_df_pd[\"spec2_path\"].values, np.zeros(len(test_pd), dtype=int), repeat=False, shuffle=False, batch_size=CFG.batch_size)\n\n# Predict\npreds = model.predict(test_ds)\n\n# Submission\ntarget_cols = [x.lower() + \"_vote\" for x in CFG.class_names]\npred_df = test_df_pd[[\"eeg_id\"]].copy()\npred_df[target_cols] = preds.tolist()\n\nsub = pd.read_csv(f\"{BASE_PATH}/sample_submission.csv\")\nsub = sub[[\"eeg_id\"]].merge(pred_df, on=\"eeg_id\", how=\"left\")\nsub.to_csv(\"submission.csv\", index=False)\nsub.head()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n\n\nLet me know if you want:\n\n* Model evaluation plots\n* Visualizations (e.g., confusion matrix, accuracy/loss curves)\n* Optimization tips\n* Help uploading the submission to Kaggle\n\nShall I help you with those next?\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"✅ Step 6: Configuration & Reproducibility\n\nimport keras\nimport tensorflow as tf\nimport numpy as np\nimport os\n\nclass CFG:\n    seed = 42\n    verbose = 1\n    preset = \"efficientnetv2_b2_imagenet\"\n    image_size = [400, 300]\n    epochs = 13\n    batch_size = 64\n    lr_mode = \"cos\"\n    drop_remainder = True\n    num_classes = 6\n    fold = 0\n    class_names = ['Seizure', 'LPD', 'GPD', 'LRDA','GRDA', 'Other']\n    label2name = dict(enumerate(class_names))\n    name2label = {v: k for k, v in label2name.items()}\n\n# Set seed\nkeras.utils.set_random_seed(CFG.seed)\n\n✅ Step 7: Stratified Fold Split\nfrom sklearn.model_selection import StratifiedGroupKFold\n\ntrain_df_pd[\"fold\"] = -1\nsgkf = StratifiedGroupKFold(n_splits=5, shuffle=True, random_state=CFG.seed)\n\nfor fold, (train_idx, val_idx) in enumerate(\n    sgkf.split(train_df_pd, y=train_df_pd[\"class_label\"], groups=train_df_pd[\"patient_id\"])):\n    train_df_pd.loc[val_idx, \"fold\"] = fold\n✅ Step 8: Build Dataset\npython\nCopy\nEdit\ndef decode_signal(path, offset):\n    sig = np.load(path.decode()).astype(np.float32)\n    start = offset * 2\n    pad_size = max(300 - sig.shape[1], 0)\n    if pad_size > 0:\n        sig = np.pad(sig, ((0, 0), (0, pad_size)), mode='constant')\n    sig = sig[:, :300]\n    sig = sig.T\n    sig = np.clip(sig, np.exp(-4.0), np.exp(8.0))\n    sig = np.log(sig)\n    sig = (sig - np.mean(sig)) / (np.std(sig) + 1e-6)\n    sig = np.tile(sig[..., None], [1, 1, 3])\n    return sig\n\ndef tf_decode_signal(path, offset):\n    sig = tf.numpy_function(decode_signal, [path, offset], tf.float32)\n    sig.set_shape([400, 300, 3])\n    return sig\n\ndef tf_decode_label(label):\n    label = tf.one_hot(label, CFG.num_classes)\n    return tf.cast(label, tf.float32)\n\ndef build_dataset(paths, offsets, labels=None, augment=False, repeat=True, batch_size=32, shuffle=True):\n    AUTO = tf.data.AUTOTUNE\n    if labels is not None:\n        ds = tf.data.Dataset.from_tensor_slices((paths, offsets, labels))\n        ds = ds.map(lambda x, y, z: (tf_decode_signal(x, y), tf_decode_label(z)), num_parallel_calls=AUTO)\n    else:\n        ds = tf.data.Dataset.from_tensor_slices((paths, offsets))\n        ds = ds.map(lambda x, y: tf_decode_signal(x, y), num_parallel_calls=AUTO)\n    \n    if repeat: ds = ds.repeat()\n    if shuffle: ds = ds.shuffle(1024, seed=CFG.seed)\n    ds = ds.batch(batch_size).prefetch(AUTO)\n    return ds\n✅ Step 9: Prepare Train & Valid Sets\npython\nCopy\nEdit\nsample_df = train_df_pd.groupby(\"spectrogram_id\").head(1).reset_index(drop=True)\ntrain_df = sample_df[sample_df.fold != CFG.fold]\nvalid_df = sample_df[sample_df.fold == CFG.fold]\n\ntrain_ds = build_dataset(\n    train_df[\"spec2_path\"].values, \n    train_df[\"spectrogram_label_offset_seconds\"].astype(int).values, \n    train_df[\"class_label\"].values, \n    batch_size=CFG.batch_size\n)\n\nvalid_ds = build_dataset(\n    valid_df[\"spec2_path\"].values, \n    valid_df[\"spectrogram_label_offset_seconds\"].astype(int).values, \n    valid_df[\"class_label\"].values, \n    repeat=False, shuffle=False,\n    batch_size=CFG.batch_size\n)\n✅ Step 10: Visualize Dataset\n\nimport matplotlib.pyplot as plt\n\nimgs, tars = next(iter(train_ds))\nplt.figure(figsize=(12, 10))\nfor i in range(8):\n    plt.subplot(2, 4, i+1)\n    img = imgs[i].numpy()[..., 0]\n    plt.imshow(img, cmap='viridis')\n    plt.title(f\"Class: {CFG.label2name[np.argmax(tars[i].numpy())]}\")\n    plt.axis(\"off\")\nplt.tight_layout()\nplt.show()\n✅ Step 11: Model, Loss & Optimizer\npython\nCopy\nEdit\nimport keras_cv\n\nmodel = keras_cv.models.ImageClassifier.from_preset(CFG.preset, num_classes=CFG.num_classes)\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(1e-4),\n    loss=tf.keras.losses.KLDivergence()\n)\n✅ Step 12: Learning Rate Schedule\npython\nCopy\nEdit\nimport math\n\ndef get_lr_callback():\n    def lrfn(epoch):\n        lr_start, lr_max, lr_min = 5e-5, 6e-6 * CFG.batch_size, 1e-5\n        lr_ramp_ep, lr_sus_ep, lr_decay = 3, 0, 0.75\n        if epoch < lr_ramp_ep: return (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n        elif epoch < lr_ramp_ep + lr_sus_ep: return lr_max\n        else:\n            decay_total = CFG.epochs - lr_ramp_ep - lr_sus_ep + 3\n            phase = math.pi * (epoch - lr_ramp_ep - lr_sus_ep) / decay_total\n            return (lr_max - lr_min) * 0.5 * (1 + math.cos(phase)) + lr_min\n    return tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=0)\n\nlr_cb = get_lr_callback()\nckpt_cb = tf.keras.callbacks.ModelCheckpoint(\"best_model.keras\", monitor=\"val_loss\", save_best_only=True)\n✅ Step 13: Training\npython\nCopy\nEdit\nhistory = model.fit(\n    train_ds,\n    validation_data=valid_ds,\n    steps_per_epoch=len(train_df) // CFG.batch_size,\n    epochs=CFG.epochs,\n    callbacks=[lr_cb, ckpt_cb],\n    verbose=CFG.verbose\n)\n✅ Step 14: Inference & Submission\n\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}