{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport math\nimport numpy as np\nimport polars as pl\nimport tensorflow as tf\nimport albumentations as A\nimport cv2\nimport torch\nimport torchaudio\nfrom scipy.signal import butter, filtfilt\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T16:38:04.761051Z","iopub.execute_input":"2025-04-22T16:38:04.761285Z","iopub.status.idle":"2025-04-22T16:38:37.999937Z","shell.execute_reply.started":"2025-04-22T16:38:04.761259Z","shell.execute_reply":"2025-04-22T16:38:37.999375Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification\"\nTRAIN_EEG_PATH = os.path.join(BASE_PATH, \"train_eegs\")\nTEST_EEG_PATH = os.path.join(BASE_PATH, \"test_eegs\")\n\nn_fft = 800\nwin_length = 256\nhop_length = 44\n\nspec_transform = torchaudio.transforms.Spectrogram(\n    n_fft=n_fft, win_length=win_length, hop_length=hop_length, power=None\n)\n\nspec_transforms = A.Compose([\n    A.Resize(height=96, width=224, interpolation=cv2.INTER_CUBIC, always_apply=True)\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T16:38:44.149436Z","iopub.execute_input":"2025-04-22T16:38:44.149728Z","iopub.status.idle":"2025-04-22T16:38:44.272410Z","shell.execute_reply.started":"2025-04-22T16:38:44.149705Z","shell.execute_reply":"2025-04-22T16:38:44.271616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def MAD(signal, axis=-1):\n    median = np.median(signal, axis=axis, keepdims=True)\n    abs_dev = np.abs(signal - median)\n    mad = np.median(abs_dev, axis=axis, keepdims=True)\n    return mad * 1.4826\n\ndef butter_filter(data, fs=200, cutoff_freq=[0.25, 50], order=5, btype=\"bandpass\"):\n    nyq = 0.5 * fs\n    low = cutoff_freq[0] / nyq\n    high = cutoff_freq[1] / nyq\n    b, a = butter(order, [low, high], btype=btype)\n    return filtfilt(b, a, data)\n\nimport math\nimport numpy as np\n\ndef bin_array(array, bin_size=4, axis=-1):\n    #print(f\"Original shape: {array.shape}\")\n    \n    length = array.shape[axis]\n    \n    # Calculate padding length, ensure it's only added when necessary\n    pad_len = (math.ceil(length / bin_size) * bin_size) - length\n    #print(f\"Padding length: {pad_len}\")\n    \n    # Pad the array to align it with bin_size\n    if pad_len > 0:\n        array = np.pad(array, [(0, 0)] * axis + [(0, pad_len)] + [(0, 0)] * (array.ndim - axis - 1), mode=\"reflect\")\n    #print(f\"Padded shape: {array.shape}\")\n    \n    # Calculate the new shape after binning\n    num_bins = array.shape[axis] // bin_size  # Number of bins\n    new_shape = list(array.shape)\n    new_shape[axis] = num_bins  # The first dimension becomes the number of bins\n    new_shape.insert(axis + 1, bin_size)  # The second dimension corresponds to bin_size\n    #print(f\"New shape after insertion: {new_shape}\")\n    \n    # Reshape and compute the mean across the bins\n    reshaped_array = array.reshape(new_shape).mean(axis=axis + 1)  # Take mean across the bin_size dimension\n    #print(f\"Reshaped array shape: {reshaped_array.shape}\")\n    \n    return reshaped_array\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T16:38:48.998920Z","iopub.execute_input":"2025-04-22T16:38:48.999551Z","iopub.status.idle":"2025-04-22T16:38:49.006312Z","shell.execute_reply.started":"2025-04-22T16:38:48.999526Z","shell.execute_reply":"2025-04-22T16:38:49.005708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def safe_reshape_eeg(eeg, target_shape=(19, 2500)):\n    \"\"\"Safely reshape EEG array to (19, 2500) by trimming or padding if needed.\"\"\"\n    total_channels = target_shape[0]\n    target_length = target_shape[1]\n    expected_size = total_channels * target_length\n\n    current_size = eeg.shape[0] * eeg.shape[1]\n    if current_size < expected_size:\n        # Pad at the end with reflection\n        pad_size = expected_size - current_size\n        eeg = np.pad(eeg, [(0, 0), (0, pad_size)], mode='reflect')\n    elif current_size > expected_size:\n        # Trim at the end\n        eeg = eeg[:, :expected_size // eeg.shape[0]]\n    return eeg.reshape(target_shape)\n\ndef compute_eeg_chain(df):\n    cols = [\"Fp1\",\"Fp2\",\"Fz\",\"Cz\",\"Pz\",\"F3\",\"F4\",\"F7\",\"F8\",\"C3\",\"C4\",\"P3\",\"P4\",\"T3\",\"T4\",\"T5\",\"T6\",\"O1\",\"O2\"]\n    eeg = [df[col].to_numpy() for col in cols]\n    \n    ekg = butter_filter(df[\"EKG\"].to_numpy(), cutoff_freq=[0.5, 20.0])\n    ekg = bin_array(ekg).reshape(1, -1)\n\n    def pair(a, b): return bin_array(butter_filter(a - b))\n    \n    ll = [pair(eeg[0], eeg[7]), pair(eeg[7], eeg[13]), pair(eeg[13], eeg[15]), pair(eeg[15], eeg[17])]\n    lp = [pair(eeg[0], eeg[5]), pair(eeg[5], eeg[9]), pair(eeg[9], eeg[11]), pair(eeg[11], eeg[17])]\n    rp = [pair(eeg[1], eeg[6]), pair(eeg[6], eeg[10]), pair(eeg[10], eeg[12]), pair(eeg[12], eeg[18])]\n    rl = [pair(eeg[1], eeg[8]), pair(eeg[8], eeg[14]), pair(eeg[14], eeg[16]), pair(eeg[16], eeg[18])]\n    mid = [pair(eeg[2], eeg[3]), pair(eeg[3], eeg[4])]\n\n    chains = np.stack([ll, lp, rp, rl])\n    mid = np.stack(mid)\n    \n    return chains, mid, ekg\n\ndef proc_eeg(eeg, mid, ekg):\n    eeg[np.isnan(eeg) | np.isinf(eeg)] = 0\n    mid[np.isnan(mid) | np.isinf(mid)] = 0\n    ekg[np.isnan(ekg) | np.isinf(ekg)] = 0\n\n    eeg -= eeg.mean(axis=-1, keepdims=True)\n    mid -= mid.mean(axis=-1, keepdims=True)\n\n    std = np.median(MAD(eeg, axis=-1)) + 1e-5\n    eeg = np.clip(eeg / std, -10, 10)\n    mid = np.clip(mid / std, -10, 10)\n    ekg = ekg / (MAD(ekg, axis=-1).mean() + 1e-5)\n\n    eeg = eeg.reshape(16, -1)\n    eeg = np.concatenate([eeg, mid, ekg], axis=0)\n\n    eeg = safe_reshape_eeg(eeg, target_shape=(19, 2500))  # 👈 Replace old reshape line\n\n    return eeg\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T16:38:56.318701Z","iopub.execute_input":"2025-04-22T16:38:56.319254Z","iopub.status.idle":"2025-04-22T16:38:56.329832Z","shell.execute_reply.started":"2025-04-22T16:38:56.319223Z","shell.execute_reply":"2025-04-22T16:38:56.329119Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@torch.no_grad()\ndef compute_spec(signal):\n    signal = torch.tensor(signal, dtype=torch.float32)\n    spec = spec_transform(signal)\n    spec = spec[:, :, 2:98]\n    spec = torch.abs(spec) / 15\n    spec = torch.log(spec.clip(math.exp(-4), math.exp(7)))\n    spec = spec.mean(dim=1)\n    return spec.numpy()\n\ndef compute_spec_eeg(a, b):\n    return butter_filter(a - b, cutoff_freq=[0.25, 40], order=5)\n\ndef compute_spec_chain(df):\n    eeg = [df[col].to_numpy() for col in [\"Fp1\",\"Fp2\",\"Fz\",\"Cz\",\"Pz\",\"F3\",\"F4\",\"F7\",\"F8\",\"C3\",\"C4\",\"P3\",\"P4\",\"T3\",\"T4\",\"T5\",\"T6\",\"O1\",\"O2\"]]\n    \n    def pair(a, b): return compute_spec_eeg(a, b)\n    ll = [pair(eeg[0], eeg[7]), pair(eeg[7], eeg[13]), pair(eeg[13], eeg[15]), pair(eeg[15], eeg[17])]\n    lp = [pair(eeg[0], eeg[5]), pair(eeg[5], eeg[9]), pair(eeg[9], eeg[11]), pair(eeg[11], eeg[17])]\n    rp = [pair(eeg[1], eeg[6]), pair(eeg[6], eeg[10]), pair(eeg[10], eeg[12]), pair(eeg[12], eeg[18])]\n    rl = [pair(eeg[1], eeg[8]), pair(eeg[8], eeg[14]), pair(eeg[14], eeg[16]), pair(eeg[16], eeg[18])]\n    \n    chain = np.stack([ll, lp, rp, rl])\n    chain = chain / (MAD(chain, axis=-1).mean() + 1e-5)\n    return compute_spec(chain[:, 0])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T16:39:02.538004Z","iopub.execute_input":"2025-04-22T16:39:02.538372Z","iopub.status.idle":"2025-04-22T16:39:02.552169Z","shell.execute_reply.started":"2025-04-22T16:39:02.538337Z","shell.execute_reply":"2025-04-22T16:39:02.551259Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def resolve_path(eeg_id, mode=\"train\"):\n    subdir = TRAIN_EEG_PATH if mode == \"train\" else TEST_EEG_PATH\n    return os.path.join(subdir, f\"{eeg_id}.parquet\")\n\ndef compute_spec_from_file(path):\n    try:\n        df = pl.read_parquet(path).fill_null(0)\n        return compute_spec_chain(df)\n    except Exception as e:\n        print(f\"Failed to read or process spec from {path}: {e}\")\n        return None\n\ndef compute_eeg_from_file(path):\n    try:\n        df = pl.read_parquet(path).fill_null(0)\n        return compute_eeg_chain(df)\n    except Exception as e:\n        print(f\"Failed to read or process EEG from {path}: {e}\")\n        return None\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T16:39:06.892620Z","iopub.execute_input":"2025-04-22T16:39:06.892895Z","iopub.status.idle":"2025-04-22T16:39:06.898144Z","shell.execute_reply.started":"2025-04-22T16:39:06.892873Z","shell.execute_reply":"2025-04-22T16:39:06.897412Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def proc_kspec(x):\n    # Slice the data (based on your specific needs)\n    x = x[:, 2:98]\n    \n    # Handle NaN and infinite values\n    x[np.isnan(x) | np.isinf(x)] = 0\n    \n    # Apply log transform (clip to avoid extremely small or large values)\n    x = np.log(np.clip(x, np.exp(-4), np.exp(7)))\n    \n    # Normalize along the correct axis (axis=1 for channels)\n    x = (x - x.mean(axis=1, keepdims=True)) / (x.std(axis=1, keepdims=True) + 1e-5)\n    \n    # Apply any additional spec transformations\n    x = spec_transforms(image=x)[\"image\"]\n    \n    # Print shape to debug\n    #print(x.shape)\n    \n    # Adjust reshape based on the actual number of elements\n    return x.reshape(4, 48, 112)  # Adjust based on the data size\n\n\n\n\ndef proc_eeg_spec(x):\n    x = x[:, :, 2:-2]\n    x[np.isnan(x) | np.isinf(x)] = 0\n    x += 1\n    return x.reshape(4, 96, 224)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T16:39:11.032486Z","iopub.execute_input":"2025-04-22T16:39:11.032758Z","iopub.status.idle":"2025-04-22T16:39:11.038206Z","shell.execute_reply.started":"2025-04-22T16:39:11.032740Z","shell.execute_reply":"2025-04-22T16:39:11.037440Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_file = \"1000913311.parquet\"\npath = os.path.join(BASE_PATH, \"train_eegs\", test_file)\n\nimport polars as pl\ndf = pl.read_parquet(path).fill_null(0)\n\nprint(\"Shape:\", df.shape)\nprint(\"Columns:\", df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T16:39:18.393774Z","iopub.execute_input":"2025-04-22T16:39:18.394431Z","iopub.status.idle":"2025-04-22T16:39:18.713869Z","shell.execute_reply.started":"2025-04-22T16:39:18.394409Z","shell.execute_reply":"2025-04-22T16:39:18.713271Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eeg, mid, ekg = compute_eeg_chain(df)\nprocessed_eeg = proc_eeg(eeg, mid, ekg)\nprint(\"EEG shape:\", processed_eeg.shape)\n\nspec = compute_spec_chain(df)\nprocessed_spec = proc_kspec(spec)\nprint(\"Spec shape:\", processed_spec.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T16:39:21.407013Z","iopub.execute_input":"2025-04-22T16:39:21.407598Z","iopub.status.idle":"2025-04-22T16:39:21.735391Z","shell.execute_reply.started":"2025-04-22T16:39:21.407560Z","shell.execute_reply":"2025-04-22T16:39:21.734824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom tqdm import tqdm\nimport polars as pl\n\n# Config\nSAVE_DIR = \"/kaggle/working/preprocessed\"\nBATCH_SIZE = 100  # Change this to control batch size\n\n# Create output directories\nos.makedirs(os.path.join(SAVE_DIR, \"eeg\"), exist_ok=True)\nos.makedirs(os.path.join(SAVE_DIR, \"spec\"), exist_ok=True)\n\n# Input EEG directory\neeg_dir = os.path.join(BASE_PATH, \"train_eegs\")\neeg_files = sorted(f for f in os.listdir(eeg_dir) if f.endswith(\".parquet\"))\n\n# Skip already processed files (resumable batches)\nalready_processed = set(f.replace(\".npy\", \"\") for f in os.listdir(os.path.join(SAVE_DIR, \"eeg\")))\neeg_files = [f for f in eeg_files if f.replace(\".parquet\", \"\") not in already_processed]\n\n# Tracking\nfailed_files = []\nsuccess_files = []\npartial_success_files = []\n\n# Process in batches\nfor i in range(0, len(eeg_files), BATCH_SIZE):\n    batch = eeg_files[i:i + BATCH_SIZE]\n    print(f\"\\n🔄 Processing batch {i // BATCH_SIZE + 1} / {(len(eeg_files) - 1) // BATCH_SIZE + 1}\")\n\n    for fname in tqdm(batch, desc=\"Processing EEGs\"):\n        eeg_id = fname.replace(\".parquet\", \"\")\n        path = os.path.join(eeg_dir, fname)\n\n        try:\n            df = pl.read_parquet(path).fill_null(0)\n            eeg_success, spec_success = False, False\n\n            # EEG Processing\n            try:\n                eeg, mid, ekg = compute_eeg_chain(df)\n                eeg_processed = proc_eeg(eeg, mid, ekg)\n                np.save(os.path.join(SAVE_DIR, \"eeg\", f\"{eeg_id}.npy\"), eeg_processed)\n                eeg_success = True\n            except Exception as e:\n                print(f\"[EEG FAIL] {eeg_id}: {e}\")\n                failed_files.append((eeg_id, \"eeg\"))\n\n            # Spectrogram Processing\n            try:\n                spec = compute_spec_chain(df)\n                spec_processed = proc_kspec(spec)\n                np.save(os.path.join(SAVE_DIR, \"spec\", f\"{eeg_id}.npy\"), spec_processed)\n                spec_success = True\n            except Exception as e:\n                print(f\"[SPEC FAIL] {eeg_id}: {e}\")\n                failed_files.append((eeg_id, \"spec\"))\n\n            # Logging result\n            if eeg_success and spec_success:\n                success_files.append(eeg_id)\n            elif eeg_success or spec_success:\n                partial_success_files.append(eeg_id)\n\n        except Exception as e:\n            print(f\"[FILE FAIL] {eeg_id}: {e}\")\n            failed_files.append((eeg_id, \"file\"))\n\n# Summary\nprint(f\"\\n✅ Fully preprocessed files: {len(success_files)}\")\nprint(f\"⚠️  Partially preprocessed files: {len(partial_success_files)}\")\nprint(f\"❌ Total failed files: {len(failed_files)}\")\n\n# Optional: Save summaries\nimport pandas as pd\n\npd.DataFrame(success_files, columns=[\"eeg_id\"]).to_csv(os.path.join(SAVE_DIR, \"success.csv\"), index=False)\npd.DataFrame(partial_success_files, columns=[\"eeg_id\"]).to_csv(os.path.join(SAVE_DIR, \"partial_success.csv\"), index=False)\npd.DataFrame(failed_files, columns=[\"eeg_id\", \"fail_type\"]).to_csv(os.path.join(SAVE_DIR, \"failures.csv\"), index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T16:39:33.833661Z","iopub.execute_input":"2025-04-22T16:39:33.834340Z","iopub.status.idle":"2025-04-22T17:00:36.802800Z","shell.execute_reply.started":"2025-04-22T16:39:33.834316Z","shell.execute_reply":"2025-04-22T17:00:36.802016Z"}},"outputs":[],"execution_count":null}]}