{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:04:01.250753Z","iopub.execute_input":"2025-05-26T15:04:01.251021Z","iopub.status.idle":"2025-05-26T15:04:52.109642Z","shell.execute_reply.started":"2025-05-26T15:04:01.251001Z","shell.execute_reply":"2025-05-26T15:04:52.108450Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport math\nimport numpy as np\nimport polars as pl\nimport tensorflow as tf\nimport albumentations as A\nimport cv2\nimport torch\nimport torchaudio\nfrom scipy.signal import butter, filtfilt\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:04:52.110855Z","iopub.execute_input":"2025-05-26T15:04:52.111189Z","iopub.status.idle":"2025-05-26T15:04:58.446925Z","shell.execute_reply.started":"2025-05-26T15:04:52.111169Z","shell.execute_reply":"2025-05-26T15:04:58.446267Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification\"\nTRAIN_EEG_PATH = os.path.join(BASE_PATH, \"train_eegs\")\nTEST_EEG_PATH = os.path.join(BASE_PATH, \"test_eegs\")\n\nn_fft = 800\nwin_length = 256\nhop_length = 44\n\nspec_transform = torchaudio.transforms.Spectrogram(\n    n_fft=n_fft, win_length=win_length, hop_length=hop_length, power=None\n)\n\nspec_transforms = A.Compose([\n    A.Resize(height=96, width=224, interpolation=cv2.INTER_CUBIC, always_apply=True)\n])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:04:58.447758Z","iopub.execute_input":"2025-05-26T15:04:58.448413Z","iopub.status.idle":"2025-05-26T15:04:58.456508Z","shell.execute_reply.started":"2025-05-26T15:04:58.448382Z","shell.execute_reply":"2025-05-26T15:04:58.455712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def MAD(signal, axis=-1):\n    median = np.median(signal, axis=axis, keepdims=True)\n    abs_dev = np.abs(signal - median)\n    mad = np.median(abs_dev, axis=axis, keepdims=True)\n    return mad * 1.4826\n\ndef butter_filter(data, fs=200, cutoff_freq=[0.25, 50], order=5, btype=\"bandpass\"):\n    nyq = 0.5 * fs\n    low = cutoff_freq[0] / nyq\n    high = cutoff_freq[1] / nyq\n    b, a = butter(order, [low, high], btype=btype)\n    return filtfilt(b, a, data)\n\nimport math\nimport numpy as np\n\ndef bin_array(array, bin_size=4, axis=-1):\n    #print(f\"Original shape: {array.shape}\")\n    \n    length = array.shape[axis]\n    \n    # Calculate padding length, ensure it's only added when necessary\n    pad_len = (math.ceil(length / bin_size) * bin_size) - length\n    #print(f\"Padding length: {pad_len}\")\n    \n    # Pad the array to align it with bin_size\n    if pad_len > 0:\n        array = np.pad(array, [(0, 0)] * axis + [(0, pad_len)] + [(0, 0)] * (array.ndim - axis - 1), mode=\"reflect\")\n    #print(f\"Padded shape: {array.shape}\")\n    \n    # Calculate the new shape after binning\n    num_bins = array.shape[axis] // bin_size  # Number of bins\n    new_shape = list(array.shape)\n    new_shape[axis] = num_bins  # The first dimension becomes the number of bins\n    new_shape.insert(axis + 1, bin_size)  # The second dimension corresponds to bin_size\n    #print(f\"New shape after insertion: {new_shape}\")\n    \n    # Reshape and compute the mean across the bins\n    reshaped_array = array.reshape(new_shape).mean(axis=axis + 1)  # Take mean across the bin_size dimension\n    #print(f\"Reshaped array shape: {reshaped_array.shape}\")\n    \n    return reshaped_array\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:04:58.457447Z","iopub.execute_input":"2025-05-26T15:04:58.457713Z","iopub.status.idle":"2025-05-26T15:04:58.474412Z","shell.execute_reply.started":"2025-05-26T15:04:58.457685Z","shell.execute_reply":"2025-05-26T15:04:58.473641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def safe_reshape_eeg(eeg, target_shape=(19, 2500)):\n    \"\"\"Safely reshape EEG array to (19, 2500) by trimming or padding if needed.\"\"\"\n    total_channels = target_shape[0]\n    target_length = target_shape[1]\n    expected_size = total_channels * target_length\n\n    current_size = eeg.shape[0] * eeg.shape[1]\n    if current_size < expected_size:\n        # Pad at the end with reflection\n        pad_size = expected_size - current_size\n        eeg = np.pad(eeg, [(0, 0), (0, pad_size)], mode='reflect')\n    elif current_size > expected_size:\n        # Trim at the end\n        eeg = eeg[:, :expected_size // eeg.shape[0]]\n    return eeg.reshape(target_shape)\n\ndef compute_eeg_chain(df):\n    cols = [\"Fp1\",\"Fp2\",\"Fz\",\"Cz\",\"Pz\",\"F3\",\"F4\",\"F7\",\"F8\",\"C3\",\"C4\",\"P3\",\"P4\",\"T3\",\"T4\",\"T5\",\"T6\",\"O1\",\"O2\"]\n    eeg = [df[col].to_numpy() for col in cols]\n    \n    ekg = butter_filter(df[\"EKG\"].to_numpy(), cutoff_freq=[0.5, 20.0])\n    ekg = bin_array(ekg).reshape(1, -1)\n\n    def pair(a, b): return bin_array(butter_filter(a - b))\n    \n    ll = [pair(eeg[0], eeg[7]), pair(eeg[7], eeg[13]), pair(eeg[13], eeg[15]), pair(eeg[15], eeg[17])]\n    lp = [pair(eeg[0], eeg[5]), pair(eeg[5], eeg[9]), pair(eeg[9], eeg[11]), pair(eeg[11], eeg[17])]\n    rp = [pair(eeg[1], eeg[6]), pair(eeg[6], eeg[10]), pair(eeg[10], eeg[12]), pair(eeg[12], eeg[18])]\n    rl = [pair(eeg[1], eeg[8]), pair(eeg[8], eeg[14]), pair(eeg[14], eeg[16]), pair(eeg[16], eeg[18])]\n    mid = [pair(eeg[2], eeg[3]), pair(eeg[3], eeg[4])]\n\n    chains = np.stack([ll, lp, rp, rl])\n    mid = np.stack(mid)\n    \n    return chains, mid, ekg\n\ndef proc_eeg(eeg, mid, ekg):\n    eeg[np.isnan(eeg) | np.isinf(eeg)] = 0\n    mid[np.isnan(mid) | np.isinf(mid)] = 0\n    ekg[np.isnan(ekg) | np.isinf(ekg)] = 0\n\n    eeg -= eeg.mean(axis=-1, keepdims=True)\n    mid -= mid.mean(axis=-1, keepdims=True)\n\n    std = np.median(MAD(eeg, axis=-1)) + 1e-5\n    eeg = np.clip(eeg / std, -10, 10)\n    mid = np.clip(mid / std, -10, 10)\n    ekg = ekg / (MAD(ekg, axis=-1).mean() + 1e-5)\n\n    eeg = eeg.reshape(16, -1)\n    eeg = np.concatenate([eeg, mid, ekg], axis=0)\n\n    eeg = safe_reshape_eeg(eeg, target_shape=(19, 2500))  # 👈 Replace old reshape line\n\n    return eeg\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:04:58.476474Z","iopub.execute_input":"2025-05-26T15:04:58.476739Z","iopub.status.idle":"2025-05-26T15:04:58.499416Z","shell.execute_reply.started":"2025-05-26T15:04:58.476722Z","shell.execute_reply":"2025-05-26T15:04:58.498686Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"@torch.no_grad()\ndef compute_spec(signal):\n    signal = torch.tensor(signal, dtype=torch.float32)\n    spec = spec_transform(signal)\n    spec = spec[:, :, 2:98]\n    spec = torch.abs(spec) / 15\n    spec = torch.log(spec.clip(math.exp(-4), math.exp(7)))\n    spec = spec.mean(dim=1)\n    return spec.numpy()\n\ndef compute_spec_eeg(a, b):\n    return butter_filter(a - b, cutoff_freq=[0.25, 40], order=5)\n\ndef compute_spec_chain(df):\n    eeg = [df[col].to_numpy() for col in [\"Fp1\",\"Fp2\",\"Fz\",\"Cz\",\"Pz\",\"F3\",\"F4\",\"F7\",\"F8\",\"C3\",\"C4\",\"P3\",\"P4\",\"T3\",\"T4\",\"T5\",\"T6\",\"O1\",\"O2\"]]\n    \n    def pair(a, b): return compute_spec_eeg(a, b)\n    ll = [pair(eeg[0], eeg[7]), pair(eeg[7], eeg[13]), pair(eeg[13], eeg[15]), pair(eeg[15], eeg[17])]\n    lp = [pair(eeg[0], eeg[5]), pair(eeg[5], eeg[9]), pair(eeg[9], eeg[11]), pair(eeg[11], eeg[17])]\n    rp = [pair(eeg[1], eeg[6]), pair(eeg[6], eeg[10]), pair(eeg[10], eeg[12]), pair(eeg[12], eeg[18])]\n    rl = [pair(eeg[1], eeg[8]), pair(eeg[8], eeg[14]), pair(eeg[14], eeg[16]), pair(eeg[16], eeg[18])]\n    \n    chain = np.stack([ll, lp, rp, rl])\n    chain = chain / (MAD(chain, axis=-1).mean() + 1e-5)\n    return compute_spec(chain[:, 0])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:04:58.500287Z","iopub.execute_input":"2025-05-26T15:04:58.500521Z","iopub.status.idle":"2025-05-26T15:04:58.519372Z","shell.execute_reply.started":"2025-05-26T15:04:58.500504Z","shell.execute_reply":"2025-05-26T15:04:58.518629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def resolve_path(eeg_id, mode=\"train\"):\n    subdir = TRAIN_EEG_PATH if mode == \"train\" else TEST_EEG_PATH\n    return os.path.join(subdir, f\"{eeg_id}.parquet\")\n\ndef compute_spec_from_file(path):\n    try:\n        df = pl.read_parquet(path).fill_null(0)\n        return compute_spec_chain(df)\n    except Exception as e:\n        print(f\"Failed to read or process spec from {path}: {e}\")\n        return None\n\ndef compute_eeg_from_file(path):\n    try:\n        df = pl.read_parquet(path).fill_null(0)\n        return compute_eeg_chain(df)\n    except Exception as e:\n        print(f\"Failed to read or process EEG from {path}: {e}\")\n        return None\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:04:58.520403Z","iopub.execute_input":"2025-05-26T15:04:58.520673Z","iopub.status.idle":"2025-05-26T15:04:58.537257Z","shell.execute_reply.started":"2025-05-26T15:04:58.520651Z","shell.execute_reply":"2025-05-26T15:04:58.536689Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def proc_kspec(x):\n    # Slice the data (based on your specific needs)\n    x = x[:, 2:98]\n    \n    # Handle NaN and infinite values\n    x[np.isnan(x) | np.isinf(x)] = 0\n    \n    # Apply log transform (clip to avoid extremely small or large values)\n    x = np.log(np.clip(x, np.exp(-4), np.exp(7)))\n    \n    # Normalize along the correct axis (axis=1 for channels)\n    x = (x - x.mean(axis=1, keepdims=True)) / (x.std(axis=1, keepdims=True) + 1e-5)\n    \n    # Apply any additional spec transformations\n    x = spec_transforms(image=x)[\"image\"]\n    \n    # Print shape to debug\n    #print(x.shape)\n    \n    # Adjust reshape based on the actual number of elements\n    return x.reshape(4, 48, 112)  # Adjust based on the data size\n\n\n\n\ndef proc_eeg_spec(x):\n    x = x[:, :, 2:-2]\n    x[np.isnan(x) | np.isinf(x)] = 0\n    x += 1\n    return x.reshape(4, 96, 224)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:04:58.538008Z","iopub.execute_input":"2025-05-26T15:04:58.538286Z","iopub.status.idle":"2025-05-26T15:04:58.554858Z","shell.execute_reply.started":"2025-05-26T15:04:58.538259Z","shell.execute_reply":"2025-05-26T15:04:58.554037Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_file = \"1000913311.parquet\"\npath = os.path.join(BASE_PATH, \"train_eegs\", test_file)\n\nimport polars as pl\ndf = pl.read_parquet(path).fill_null(0)\n\nprint(\"Shape:\", df.shape)\nprint(\"Columns:\", df.columns)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:04:58.555625Z","iopub.execute_input":"2025-05-26T15:04:58.555867Z","iopub.status.idle":"2025-05-26T15:04:58.589013Z","shell.execute_reply.started":"2025-05-26T15:04:58.555846Z","shell.execute_reply":"2025-05-26T15:04:58.588389Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eeg, mid, ekg = compute_eeg_chain(df)\nprocessed_eeg = proc_eeg(eeg, mid, ekg)\nprint(\"EEG shape:\", processed_eeg.shape)\n\nspec = compute_spec_chain(df)\nprocessed_spec = proc_kspec(spec)\nprint(\"Spec shape:\", processed_spec.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:04:58.589868Z","iopub.execute_input":"2025-05-26T15:04:58.590078Z","iopub.status.idle":"2025-05-26T15:04:58.645925Z","shell.execute_reply.started":"2025-05-26T15:04:58.590059Z","shell.execute_reply":"2025-05-26T15:04:58.645287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nfrom tqdm import tqdm\nimport polars as pl\n\n# Config\nSAVE_DIR = \"/kaggle/working/preprocessed\"\nBATCH_SIZE = 100  # Change this to control batch size\n\n# Create output directories\nos.makedirs(os.path.join(SAVE_DIR, \"eeg\"), exist_ok=True)\nos.makedirs(os.path.join(SAVE_DIR, \"spec\"), exist_ok=True)\n\n# Input EEG directory\neeg_dir = os.path.join(BASE_PATH, \"train_eegs\")\neeg_files = sorted(f for f in os.listdir(eeg_dir) if f.endswith(\".parquet\"))\n\n# Skip already processed files (resumable batches)\nalready_processed = set(f.replace(\".npy\", \"\") for f in os.listdir(os.path.join(SAVE_DIR, \"eeg\")))\neeg_files = [f for f in eeg_files if f.replace(\".parquet\", \"\") not in already_processed]\n\n# Tracking\nfailed_files = []\nsuccess_files = []\npartial_success_files = []\n\n# Process in batches\nfor i in range(0, len(eeg_files), BATCH_SIZE):\n    batch = eeg_files[i:i + BATCH_SIZE]\n    print(f\"\\n🔄 Processing batch {i // BATCH_SIZE + 1} / {(len(eeg_files) - 1) // BATCH_SIZE + 1}\")\n\n    for fname in tqdm(batch, desc=\"Processing EEGs\"):\n        eeg_id = fname.replace(\".parquet\", \"\")\n        path = os.path.join(eeg_dir, fname)\n\n        try:\n            df = pl.read_parquet(path).fill_null(0)\n            eeg_success, spec_success = False, False\n\n            # EEG Processing\n            try:\n                eeg, mid, ekg = compute_eeg_chain(df)\n                eeg_processed = proc_eeg(eeg, mid, ekg)\n                np.save(os.path.join(SAVE_DIR, \"eeg\", f\"{eeg_id}.npy\"), eeg_processed)\n                eeg_success = True\n            except Exception as e:\n                print(f\"[EEG FAIL] {eeg_id}: {e}\")\n                failed_files.append((eeg_id, \"eeg\"))\n\n            # Spectrogram Processing\n            try:\n                spec = compute_spec_chain(df)\n                spec_processed = proc_kspec(spec)\n                np.save(os.path.join(SAVE_DIR, \"spec\", f\"{eeg_id}.npy\"), spec_processed)\n                spec_success = True\n            except Exception as e:\n                print(f\"[SPEC FAIL] {eeg_id}: {e}\")\n                failed_files.append((eeg_id, \"spec\"))\n\n            # Logging result\n            if eeg_success and spec_success:\n                success_files.append(eeg_id)\n            elif eeg_success or spec_success:\n                partial_success_files.append(eeg_id)\n\n        except Exception as e:\n            print(f\"[FILE FAIL] {eeg_id}: {e}\")\n            failed_files.append((eeg_id, \"file\"))\n\n# Summary\nprint(f\"\\n✅ Fully preprocessed files: {len(success_files)}\")\nprint(f\"⚠️  Partially preprocessed files: {len(partial_success_files)}\")\nprint(f\"❌ Total failed files: {len(failed_files)}\")\n\n# Optional: Save summaries\nimport pandas as pd\n\npd.DataFrame(success_files, columns=[\"eeg_id\"]).to_csv(os.path.join(SAVE_DIR, \"success.csv\"), index=False)\npd.DataFrame(partial_success_files, columns=[\"eeg_id\"]).to_csv(os.path.join(SAVE_DIR, \"partial_success.csv\"), index=False)\npd.DataFrame(failed_files, columns=[\"eeg_id\", \"fail_type\"]).to_csv(os.path.join(SAVE_DIR, \"failures.csv\"), index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:04:58.646714Z","iopub.execute_input":"2025-05-26T15:04:58.647072Z","iopub.status.idle":"2025-05-26T15:24:18.598580Z","shell.execute_reply.started":"2025-05-26T15:04:58.647050Z","shell.execute_reply":"2025-05-26T15:24:18.597703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.svm import SVC\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report\nfrom sklearn.decomposition import PCA\nimport os\n\n# Load metadata\ntrain_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Directory where you saved preprocessed data\nPREPROCESSED_DIR = \"/kaggle/working/preprocessed\"\n\n# Function to load numpy arrays\ndef load_preprocessed_data(eeg_id):\n    eeg_path = os.path.join(PREPROCESSED_DIR, \"eeg\", f\"{eeg_id}.npy\")\n    spec_path = os.path.join(PREPROCESSED_DIR, \"spec\", f\"{eeg_id}.npy\")\n    \n    eeg = np.load(eeg_path) if os.path.exists(eeg_path) else None\n    spec = np.load(spec_path) if os.path.exists(spec_path) else None\n    \n    return eeg, spec\n\n# Get unique EEG IDs from metadata\neeg_ids = train_df['eeg_id'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:24:18.599445Z","iopub.execute_input":"2025-05-26T15:24:18.599720Z","iopub.status.idle":"2025-05-26T15:24:19.222179Z","shell.execute_reply.started":"2025-05-26T15:24:18.599692Z","shell.execute_reply":"2025-05-26T15:24:19.221593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize lists to store features and labels\nX = []\ny = []\n\n# Define the target columns (6 seizure types)\ntarget_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\nfor eeg_id in eeg_ids[:2000]:  # Using first 2000 samples for demo (adjust as needed)\n    eeg, spec = load_preprocessed_data(eeg_id)\n    \n    if eeg is not None and spec is not None:\n        # Flatten and concatenate features\n        eeg_flat = eeg.flatten()\n        spec_flat = spec.flatten()\n        features = np.concatenate([eeg_flat, spec_flat])\n        \n        # Get corresponding labels from metadata\n        labels = train_df[train_df['eeg_id'] == eeg_id][target_cols].values.mean(axis=0)\n        pred_class = np.argmax(labels)  # Convert to single class label\n        \n        X.append(features)\n        y.append(pred_class)\n\n# Convert to numpy arrays\nX = np.array(X)\ny = np.array(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:24:19.222851Z","iopub.execute_input":"2025-05-26T15:24:19.223055Z","iopub.status.idle":"2025-05-26T15:24:26.884638Z","shell.execute_reply.started":"2025-05-26T15:24:19.223040Z","shell.execute_reply":"2025-05-26T15:24:26.883999Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Standardize features first\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\n# Apply PCA to reduce dimensions\npca = PCA(n_components=100)  # Adjust based on your needs\nX_pca = pca.fit_transform(X_scaled)\n\nprint(f\"Explained variance ratio: {sum(pca.explained_variance_ratio_):.2f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:24:26.886820Z","iopub.execute_input":"2025-05-26T15:24:26.887072Z","iopub.status.idle":"2025-05-26T15:24:41.489618Z","shell.execute_reply.started":"2025-05-26T15:24:26.887054Z","shell.execute_reply":"2025-05-26T15:24:41.488867Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(\n    X_pca, y, test_size=0.2, random_state=42, stratify=y\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:24:41.490399Z","iopub.execute_input":"2025-05-26T15:24:41.490818Z","iopub.status.idle":"2025-05-26T15:24:41.498293Z","shell.execute_reply.started":"2025-05-26T15:24:41.490789Z","shell.execute_reply":"2025-05-26T15:24:41.497704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Initialize SVM - using linear kernel for efficiency\nsvm = SVC(kernel='linear', C=1.0, random_state=42, class_weight='balanced')\n\n# Train the model\nsvm.fit(X_train, y_train)\n\n# Evaluate\ntrain_score = svm.score(X_train, y_train)\ntest_score = svm.score(X_test, y_test)\n\nprint(f\"Training Accuracy: {train_score:.2f}\")\nprint(f\"Test Accuracy: {test_score:.2f}\")\n\n# Detailed classification report\ny_pred = svm.predict(X_test)\nprint(classification_report(y_test, y_pred, target_names=target_cols))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:24:41.499027Z","iopub.execute_input":"2025-05-26T15:24:41.499214Z","iopub.status.idle":"2025-05-26T15:34:57.856240Z","shell.execute_reply.started":"2025-05-26T15:24:41.499194Z","shell.execute_reply":"2025-05-26T15:34:57.855397Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import joblib\n\n# Save the trained model\njoblib.dump(svm, '/kaggle/working/eeg_svm_model.pkl')\n\n# To load later:\n# svm = joblib.load('/kaggle/working/eeg_svm_model.pkl')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:34:57.857076Z","iopub.execute_input":"2025-05-26T15:34:57.857372Z","iopub.status.idle":"2025-05-26T15:34:57.865780Z","shell.execute_reply.started":"2025-05-26T15:34:57.857347Z","shell.execute_reply":"2025-05-26T15:34:57.865091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nmodel_path = '/kaggle/working/eeg_svm_model.pkl'\nprint(f\"File exists: {os.path.exists(model_path)}\")\n\n# List all files in /kaggle/working\nprint(\"Files in /kaggle/working:\")\nprint(os.listdir('/kaggle/working'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:34:57.866421Z","iopub.execute_input":"2025-05-26T15:34:57.866606Z","iopub.status.idle":"2025-05-26T15:34:57.882206Z","shell.execute_reply.started":"2025-05-26T15:34:57.866591Z","shell.execute_reply":"2025-05-26T15:34:57.881526Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.svm import SVC\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report\nfrom sklearn.decomposition import PCA\nimport os\nfrom scipy.stats import entropy  # For KL-Divergence\n\n# Load metadata\ntrain_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n\n# Directory where you saved preprocessed data\nPREPROCESSED_DIR = \"/kaggle/working/preprocessed\"\n\n# Function to load numpy arrays\ndef load_preprocessed_data(eeg_id):\n    eeg_path = os.path.join(PREPROCESSED_DIR, \"eeg\", f\"{eeg_id}.npy\")\n    spec_path = os.path.join(PREPROCESSED_DIR, \"spec\", f\"{eeg_id}.npy\")\n    \n    eeg = np.load(eeg_path) if os.path.exists(eeg_path) else None\n    spec = np.load(spec_path) if os.path.exists(spec_path) else None\n    \n    return eeg, spec\n\n# Function to calculate KL-Divergence\ndef calculate_kl_divergence(y_true, y_pred, num_classes=6):\n    \"\"\"Calculate KL divergence between true and predicted distributions\"\"\"\n    # Create probability distributions\n    true_dist = np.bincount(y_true, minlength=num_classes) / len(y_true)\n    pred_dist = np.bincount(y_pred, minlength=num_classes) / len(y_pred)\n    \n    # Add small epsilon to avoid division by zero\n    epsilon = 1e-10\n    true_dist = true_dist + epsilon\n    pred_dist = pred_dist + epsilon\n    \n    # Normalize\n    true_dist = true_dist / np.sum(true_dist)\n    pred_dist = pred_dist / np.sum(pred_dist)\n    \n    return entropy(true_dist, pred_dist)\n\n# Get unique EEG IDs from metadata\neeg_ids = train_df['eeg_id'].unique()\n\n# Initialize lists to store features and labels\nX = []\ny = []\n\n# Define the target columns (6 seizure types)\ntarget_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\nfor eeg_id in eeg_ids[:2000]:  # Using first 2000 samples for demo (adjust as needed)\n    eeg, spec = load_preprocessed_data(eeg_id)\n    \n    if eeg is not None and spec is not None:\n        # Flatten and concatenate features\n        eeg_flat = eeg.flatten()\n        spec_flat = spec.flatten()\n        features = np.concatenate([eeg_flat, spec_flat])\n        \n        # Get corresponding labels from metadata\n        labels = train_df[train_df['eeg_id'] == eeg_id][target_cols].values.mean(axis=0)\n        pred_class = np.argmax(labels)  # Convert to single class label\n        \n        X.append(features)\n        y.append(pred_class)\n\n# Convert to numpy arrays\nX = np.array(X)\ny = np.array(y)\n\n# Standardize features first\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\n# Apply PCA to reduce dimensions\npca = PCA(n_components=100)  # Adjust based on your needs\nX_pca = pca.fit_transform(X_scaled)\n\nprint(f\"Explained variance ratio: {sum(pca.explained_variance_ratio_):.2f}\")\n\nX_train, X_test, y_train, y_test = train_test_split(\n    X_pca, y, test_size=0.2, random_state=42, stratify=y\n)\n\n# Initialize SVM - using linear kernel for efficiency\nsvm = SVC(kernel='linear', C=1.0, random_state=42, class_weight='balanced')\n\n# Train the model\nsvm.fit(X_train, y_train)\n\n# Evaluate\ntrain_score = svm.score(X_train, y_train)\ntest_score = svm.score(X_test, y_test)\n\nprint(f\"Training Accuracy: {train_score:.2f}\")\nprint(f\"Test Accuracy: {test_score:.2f}\")\n\n# Detailed classification report\ny_pred = svm.predict(X_test)\nprint(classification_report(y_test, y_pred, target_names=target_cols))\n\n# Calculate KL-Divergence\nkl_div = calculate_kl_divergence(y_test, y_pred)\nprint(f\"\\nKL Divergence between true and predicted distributions: {kl_div:.4f}\")\n\n# Save the trained model\nimport joblib\njoblib.dump(svm, '/kaggle/working/eeg_svm_model.pkl')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:34:57.882948Z","iopub.execute_input":"2025-05-26T15:34:57.883614Z","iopub.status.idle":"2025-05-26T15:44:08.159428Z","shell.execute_reply.started":"2025-05-26T15:34:57.883593Z","shell.execute_reply":"2025-05-26T15:44:08.158484Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#new code with apparently higher accuracy","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import classification_report\nfrom scipy.stats import entropy\nimport joblib\n\n# Load metadata\ntrain_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nPREPROCESSED_DIR = \"/kaggle/working/preprocessed\"\n\n# Columns with seizure types\ntarget_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\n# Load preprocessed data\ndef load_preprocessed_data(eeg_id):\n    eeg_path = os.path.join(PREPROCESSED_DIR, \"eeg\", f\"{eeg_id}.npy\")\n    spec_path = os.path.join(PREPROCESSED_DIR, \"spec\", f\"{eeg_id}.npy\")\n    eeg = np.load(eeg_path) if os.path.exists(eeg_path) else None\n    spec = np.load(spec_path) if os.path.exists(spec_path) else None\n    return eeg, spec\n\n# Statistical feature extractor\ndef extract_stat_features(data):\n    stats = []\n    for channel in data:  # Assumes shape (channels, time) or (freq, time)\n        stats.extend([\n            np.mean(channel), np.std(channel), np.max(channel),\n            np.min(channel), np.median(channel),\n            np.percentile(channel, 25), np.percentile(channel, 75)\n        ])\n    return stats\n\n# KL divergence calculator\ndef calculate_kl_divergence(y_true, y_pred, num_classes=6):\n    true_dist = np.bincount(y_true, minlength=num_classes) / len(y_true)\n    pred_dist = np.bincount(y_pred, minlength=num_classes) / len(y_pred)\n    epsilon = 1e-10\n    true_dist += epsilon\n    pred_dist += epsilon\n    true_dist /= true_dist.sum()\n    pred_dist /= pred_dist.sum()\n    return entropy(true_dist, pred_dist)\n\n# Prepare data\nX, y = [], []\neeg_ids = train_df['eeg_id'].unique()\n\nfor eeg_id in eeg_ids[:2000]:  # Limit to 2000 samples\n    eeg, spec = load_preprocessed_data(eeg_id)\n    if eeg is None or spec is None:\n        continue\n\n    row = train_df[train_df['eeg_id'] == eeg_id][target_cols].mean()\n    if row.max() < 0.5:\n        continue  # Skip unclear labels\n\n    label = np.argmax(row)\n    eeg_features = extract_stat_features(eeg)\n    spec_features = extract_stat_features(spec)\n    combined_features = eeg_features + spec_features\n    X.append(combined_features)\n    y.append(label)\n\n# Convert to arrays\nX = np.array(X)\ny = np.array(y)\n\n# Scale features\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\n# PCA for dimensionality reduction\npca = PCA(n_components=0.95)  # Keep 95% of variance\nX_pca = pca.fit_transform(X_scaled)\nprint(f\"PCA - Components selected: {X_pca.shape[1]}\")\n\n# Train-test split\nX_train, X_test, y_train, y_test = train_test_split(\n    X_pca, y, test_size=0.2, random_state=42, stratify=y\n)\n\n# Train model\nclf = RandomForestClassifier(\n    n_estimators=200, max_depth=20, random_state=42, class_weight='balanced'\n)\nclf.fit(X_train, y_train)\n\n# Evaluate\ntrain_acc = clf.score(X_train, y_train)\ntest_acc = clf.score(X_test, y_test)\nprint(f\"Train Accuracy: {train_acc:.2f}\")\nprint(f\"Test Accuracy: {test_acc:.2f}\")\n\n# Classification report\ny_pred = clf.predict(X_test)\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_test, y_pred, target_names=target_cols))\n\n# KL Divergence\nkl_div = calculate_kl_divergence(y_test, y_pred)\nprint(f\"KL Divergence: {kl_div:.4f}\")\n\n# Save model\njoblib.dump(clf, '/kaggle/working/eeg_rf_model.pkl')\nprint(\"Model saved as 'eeg_rf_model.pkl'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T15:55:26.799723Z","iopub.execute_input":"2025-05-26T15:55:26.800023Z","iopub.status.idle":"2025-05-26T15:55:49.950742Z","shell.execute_reply.started":"2025-05-26T15:55:26.800002Z","shell.execute_reply":"2025-05-26T15:55:49.949893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#new code for svm above is random forest","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import classification_report\nfrom scipy.stats import entropy\nimport joblib\n\n# Load metadata\ntrain_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nPREPROCESSED_DIR = \"/kaggle/working/preprocessed\"\ntarget_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\ndef load_preprocessed_data(eeg_id):\n    eeg_path = os.path.join(PREPROCESSED_DIR, \"eeg\", f\"{eeg_id}.npy\")\n    spec_path = os.path.join(PREPROCESSED_DIR, \"spec\", f\"{eeg_id}.npy\")\n    eeg = np.load(eeg_path) if os.path.exists(eeg_path) else None\n    spec = np.load(spec_path) if os.path.exists(spec_path) else None\n    return eeg, spec\n\ndef extract_stat_features(data):\n    stats = []\n    for channel in data:\n        stats.extend([\n            np.mean(channel), np.std(channel), np.max(channel),\n            np.min(channel), np.median(channel),\n            np.percentile(channel, 25), np.percentile(channel, 75)\n        ])\n    return stats\n\ndef calculate_kl_divergence(y_true, y_pred, num_classes=6):\n    true_dist = np.bincount(y_true, minlength=num_classes) / len(y_true)\n    pred_dist = np.bincount(y_pred, minlength=num_classes) / len(y_pred)\n    epsilon = 1e-10\n    true_dist += epsilon\n    pred_dist += epsilon\n    true_dist /= true_dist.sum()\n    pred_dist /= pred_dist.sum()\n    return entropy(true_dist, pred_dist)\n\n# Prepare data\nX, y = [], []\neeg_ids = train_df['eeg_id'].unique()\n\nfor eeg_id in eeg_ids[:2000]:  # Adjust range for full training\n    eeg, spec = load_preprocessed_data(eeg_id)\n    if eeg is None or spec is None:\n        continue\n\n    row = train_df[train_df['eeg_id'] == eeg_id][target_cols].mean()\n    if row.max() < 0.5:\n        continue  # Skip low-confidence labels\n\n    label = np.argmax(row)\n    eeg_features = extract_stat_features(eeg)\n    spec_features = extract_stat_features(spec)\n    combined_features = eeg_features + spec_features\n    X.append(combined_features)\n    y.append(label)\n\nX = np.array(X)\ny = np.array(y)\n\n# Standardize\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\n# PCA (retain 95% variance)\npca = PCA(n_components=0.95)\nX_pca = pca.fit_transform(X_scaled)\n\n\n# Split\nX_train, X_test, y_train, y_test = train_test_split(\n    X_pca, y, test_size=0.2, stratify=y, random_state=42\n)\n\n# SVM with RBF kernel (higher accuracy)\nsvm = SVC(kernel='rbf', C=10, gamma='scale', class_weight='balanced', random_state=42)\nsvm.fit(X_train, y_train)\n\n# Evaluate\ntrain_acc = svm.score(X_train, y_train)\ntest_acc = svm.score(X_test, y_test)\nprint(f\"Train Accuracy: {train_acc:.2f}\")\nprint(f\"Test Accuracy: {test_acc:.2f}\")\n\n# Report\ny_pred = svm.predict(X_test)\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_test, y_pred, target_names=target_cols))\n\n# KL Divergence\nkl_div = calculate_kl_divergence(y_test, y_pred)\nprint(f\"KL Divergence: {kl_div:.4f}\")\n\n# Save model\njoblib.dump(svm, '/kaggle/working/eeg_svm_rbf_model.pkl')\nprint(\"Model saved as 'eeg_svm_rbf_model.pkl'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T16:02:32.548934Z","iopub.execute_input":"2025-05-26T16:02:32.549637Z","iopub.status.idle":"2025-05-26T16:02:53.986011Z","shell.execute_reply.started":"2025-05-26T16:02:32.549614Z","shell.execute_reply":"2025-05-26T16:02:53.985270Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#new latest","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import classification_report\nfrom scipy.stats import entropy\nimport joblib\n\n# Load metadata\ntrain_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nPREPROCESSED_DIR = \"/kaggle/working/preprocessed\"\ntarget_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\ndef load_preprocessed_data(eeg_id):\n    eeg_path = os.path.join(PREPROCESSED_DIR, \"eeg\", f\"{eeg_id}.npy\")\n    spec_path = os.path.join(PREPROCESSED_DIR, \"spec\", f\"{eeg_id}.npy\")\n    eeg = np.load(eeg_path) if os.path.exists(eeg_path) else None\n    spec = np.load(spec_path) if os.path.exists(spec_path) else None\n    return eeg, spec\n\ndef extract_stat_features(data):\n    stats = []\n    for channel in data:\n        stats.extend([\n            np.mean(channel), np.std(channel), np.max(channel),\n            np.min(channel), np.median(channel),\n            np.percentile(channel, 25), np.percentile(channel, 75)\n        ])\n    return stats\n\ndef calculate_kl_divergence(y_true, y_pred, num_classes=6):\n    true_dist = np.bincount(y_true, minlength=num_classes) / len(y_true)\n    pred_dist = np.bincount(y_pred, minlength=num_classes) / len(y_pred)\n    epsilon = 1e-10\n    true_dist += epsilon\n    pred_dist += epsilon\n    true_dist /= true_dist.sum()\n    pred_dist /= pred_dist.sum()\n    return entropy(true_dist, pred_dist)\n\n# Prepare data\nX, y = [], []\neeg_ids = train_df['eeg_id'].unique()\n\nfor eeg_id in eeg_ids:  # ✅ Use all available EEG IDs\n    eeg, spec = load_preprocessed_data(eeg_id)\n    if eeg is None or spec is None:\n        continue\n\n    row = train_df[train_df['eeg_id'] == eeg_id][target_cols].mean()\n    if row.max() < 0.5:\n        continue  # Skip low-confidence labels\n\n    label = np.argmax(row)\n    eeg_features = extract_stat_features(eeg)\n    spec_features = extract_stat_features(spec)\n    combined_features = eeg_features + spec_features\n    X.append(combined_features)\n    y.append(label)\n\nX = np.array(X)\ny = np.array(y)\n\n# Standardize\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\n# PCA (retain 95% variance)\npca = PCA(n_components=0.95)\nX_pca = pca.fit_transform(X_scaled)\n\n# Split\nX_train, X_test, y_train, y_test = train_test_split(\n    X_pca, y, test_size=0.2, stratify=y, random_state=42\n)\n\n# SVM with RBF kernel (high accuracy)\nsvm = SVC(kernel='rbf', C=10, gamma='scale', class_weight='balanced', random_state=42)\nsvm.fit(X_train, y_train)\n\n# Evaluate\ntrain_acc = svm.score(X_train, y_train)\ntest_acc = svm.score(X_test, y_test)\nprint(f\"Train Accuracy: {train_acc:.2f}\")\nprint(f\"Test Accuracy: {test_acc:.2f}\")\n\n# Report\ny_pred = svm.predict(X_test)\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_test, y_pred, target_names=target_cols))\n\n# KL Divergence\nkl_div = calculate_kl_divergence(y_test, y_pred)\nprint(f\"KL Divergence: {kl_div:.4f}\")\n\n# Save model\njoblib.dump(svm, '/kaggle/working/eeg_svm_rbf_model.pkl')\nprint(\"Model saved as 'eeg_svm_rbf_model.pkl'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T16:07:04.461897Z","iopub.execute_input":"2025-05-26T16:07:04.462550Z","iopub.status.idle":"2025-05-26T16:11:09.639206Z","shell.execute_reply.started":"2025-05-26T16:07:04.462525Z","shell.execute_reply":"2025-05-26T16:11:09.638286Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#the above code uses all data","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#the below code is random forest which uses all data","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import classification_report\nfrom scipy.stats import entropy\nimport joblib\n\n# Load metadata\ntrain_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nPREPROCESSED_DIR = \"/kaggle/working/preprocessed\"\ntarget_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\ndef load_preprocessed_data(eeg_id):\n    eeg_path = os.path.join(PREPROCESSED_DIR, \"eeg\", f\"{eeg_id}.npy\")\n    spec_path = os.path.join(PREPROCESSED_DIR, \"spec\", f\"{eeg_id}.npy\")\n    eeg = np.load(eeg_path) if os.path.exists(eeg_path) else None\n    spec = np.load(spec_path) if os.path.exists(spec_path) else None\n    return eeg, spec\n\ndef extract_stat_features(data):\n    stats = []\n    for channel in data:\n        stats.extend([\n            np.mean(channel), np.std(channel), np.max(channel),\n            np.min(channel), np.median(channel),\n            np.percentile(channel, 25), np.percentile(channel, 75)\n        ])\n    return stats\n\ndef calculate_kl_divergence(y_true, y_pred, num_classes=6):\n    true_dist = np.bincount(y_true, minlength=num_classes) / len(y_true)\n    pred_dist = np.bincount(y_pred, minlength=num_classes) / len(y_pred)\n    epsilon = 1e-10\n    true_dist += epsilon\n    pred_dist += epsilon\n    true_dist /= true_dist.sum()\n    pred_dist /= pred_dist.sum()\n    return entropy(true_dist, pred_dist)\n\n# Prepare dataset\nX, y = [], []\neeg_ids = train_df['eeg_id'].unique()\n\nfor eeg_id in eeg_ids:  # ✅ Use all available EEG IDs\n    eeg, spec = load_preprocessed_data(eeg_id)\n    if eeg is None or spec is None:\n        continue\n\n    row = train_df[train_df['eeg_id'] == eeg_id][target_cols].mean()\n    if row.max() < 0.5:\n        continue  # Skip low-confidence labels\n\n    label = np.argmax(row)\n    eeg_features = extract_stat_features(eeg)\n    spec_features = extract_stat_features(spec)\n    combined_features = eeg_features + spec_features\n    X.append(combined_features)\n    y.append(label)\n\nX = np.array(X)\ny = np.array(y)\n\n# Standardize\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\n# PCA (retain 95% variance)\npca = PCA(n_components=0.95)\nX_pca = pca.fit_transform(X_scaled)\n\n# Split data\nX_train, X_test, y_train, y_test = train_test_split(\n    X_pca, y, test_size=0.2, stratify=y, random_state=42\n)\n\n# Train Random Forest\nrf = RandomForestClassifier(\n    n_estimators=200,\n    max_depth=30,\n    class_weight='balanced',\n    random_state=42,\n    n_jobs=-1\n)\nrf.fit(X_train, y_train)\n\n# Evaluation\ntrain_acc = rf.score(X_train, y_train)\ntest_acc = rf.score(X_test, y_test)\nprint(f\"Train Accuracy: {train_acc:.2f}\")\nprint(f\"Test Accuracy: {test_acc:.2f}\")\n\n# Classification Report\ny_pred = rf.predict(X_test)\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_test, y_pred, target_names=target_cols))\n\n# KL Divergence\nkl_div = calculate_kl_divergence(y_test, y_pred)\nprint(f\"KL Divergence: {kl_div:.4f}\")\n\n# Save model\njoblib.dump(rf, '/kaggle/working/eeg_rf_model.pkl')\nprint(\"Model saved as 'eeg_rf_model.pkl'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T16:12:43.574790Z","iopub.execute_input":"2025-05-26T16:12:43.575670Z","iopub.status.idle":"2025-05-26T16:15:50.021135Z","shell.execute_reply.started":"2025-05-26T16:12:43.575646Z","shell.execute_reply":"2025-05-26T16:15:50.020437Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#sankalp code svm","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom xgboost import XGBClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nfrom sklearn.metrics import classification_report\nfrom scipy.stats import entropy\nimport joblib\n\n# Load metadata\ntrain_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nPREPROCESSED_DIR = \"/kaggle/working/preprocessed\"\ntarget_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\ndef load_preprocessed_data(eeg_id):\n    eeg_path = os.path.join(PREPROCESSED_DIR, \"eeg\", f\"{eeg_id}.npy\")\n    spec_path = os.path.join(PREPROCESSED_DIR, \"spec\", f\"{eeg_id}.npy\")\n    eeg = np.load(eeg_path) if os.path.exists(eeg_path) else None\n    spec = np.load(spec_path) if os.path.exists(spec_path) else None\n    return eeg, spec\n\ndef extract_stat_features(data):\n    stats = []\n    for channel in data:\n        stats.extend([\n            np.mean(channel), np.std(channel), np.max(channel),\n            np.min(channel), np.median(channel),\n            np.percentile(channel, 25), np.percentile(channel, 75)\n        ])\n    return stats\n\ndef calculate_kl_divergence(y_true, y_pred, num_classes=6):\n    true_dist = np.bincount(y_true, minlength=num_classes) / len(y_true)\n    pred_dist = np.bincount(y_pred, minlength=num_classes) / len(y_pred)\n    epsilon = 1e-10\n    true_dist += epsilon\n    pred_dist += epsilon\n    true_dist /= true_dist.sum()\n    pred_dist /= pred_dist.sum()\n    return entropy(true_dist, pred_dist)\n\n# Prepare dataset\nX, y = [], []\neeg_ids = train_df['eeg_id'].unique()\n\nfor eeg_id in eeg_ids:  # ✅ Use all available EEG IDs\n    eeg, spec = load_preprocessed_data(eeg_id)\n    if eeg is None or spec is None:\n        continue\n\n    row = train_df[train_df['eeg_id'] == eeg_id][target_cols].mean()\n    if row.max() < 0.5:\n        continue  # Skip low-confidence labels\n\n    label = np.argmax(row)\n    eeg_features = extract_stat_features(eeg)\n    spec_features = extract_stat_features(spec)\n    combined_features = eeg_features + spec_features\n    X.append(combined_features)\n    y.append(label)\n\nX = np.array(X)\ny = np.array(y)\n\n# Standardize\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\n# PCA (retain 95% variance)\npca = PCA(n_components=0.95)\nX_pca = pca.fit_transform(X_scaled)\n\n# Train-test split\nX_train, X_test, y_train, y_test = train_test_split(\n    X_pca, y, test_size=0.2, stratify=y, random_state=42\n)\n\n# XGBoost Classifier\nxgb = XGBClassifier(\n    n_estimators=300,\n    max_depth=8,\n    learning_rate=0.05,\n    subsample=0.9,\n    colsample_bytree=0.8,\n    use_label_encoder=False,\n    eval_metric='mlogloss',\n    random_state=42\n)\nxgb.fit(X_train, y_train)\n\n# Evaluation\ntrain_acc = xgb.score(X_train, y_train)\ntest_acc = xgb.score(X_test, y_test)\nprint(f\"Train Accuracy: {train_acc:.2f}\")\nprint(f\"Test Accuracy: {test_acc:.2f}\")\n\n# Report\ny_pred = xgb.predict(X_test)\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_test, y_pred, target_names=target_cols))\n\n# KL Divergence\nkl_div = calculate_kl_divergence(y_test, y_pred)\nprint(f\"KL Divergence: {kl_div:.4f}\")\n\n# Save model\njoblib.dump(xgb, '/kaggle/working/eeg_xgboost_model.pkl')\nprint(\"Model saved as 'eeg_xgboost_model.pkl'\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T16:32:44.692842Z","iopub.execute_input":"2025-05-26T16:32:44.693449Z","iopub.status.idle":"2025-05-26T16:36:28.619545Z","shell.execute_reply.started":"2025-05-26T16:32:44.693424Z","shell.execute_reply":"2025-05-26T16:36:28.618647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#xg boost","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold, GridSearchCV\nfrom sklearn.metrics import log_loss, confusion_matrix, ConfusionMatrixDisplay\nfrom tqdm import tqdm\nimport pickle\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport multiprocessing\nfrom xgboost import XGBClassifier\n\n# Configuration\nTEST_PATH = \"/kaggle/working/\"\nBASE_PATH = \"/kaggle/input/hms-harmful-brain-activity-classification\"\nPREPROCESSED_PATH = \"/kaggle/input/preprocessing/preprocessed\"\nTRAIN_LABELS_PATH = os.path.join(BASE_PATH, \"train.csv\")\nMODEL_OUTPUT_PATH = os.path.join(TEST_PATH, \"models\")\n\n# Define our classes\nCLASSES = ['Seizure', 'LPD', 'GPD', 'LRDA', 'GRDA', 'Other']\n\n# Metrics functions\ndef kl_divergence(y_true, y_pred):\n    \"\"\"\n    Calculate KL divergence between true and predicted probabilities\n    \n    Args:\n        y_true: One-hot encoded ground truth (N, C)\n        y_pred: Predicted probabilities (N, C)\n        \n    Returns:\n        Mean KL divergence\n    \"\"\"\n    epsilon = 1e-10\n    y_pred = np.clip(y_pred, epsilon, 1 - epsilon)\n    y_true = np.clip(y_true, epsilon, 1.0)\n    \n    kl_div = np.sum(y_true * np.log(y_true / y_pred), axis=1)\n    return np.mean(kl_div)\n\ndef kl_divergence_scorer(estimator, X, y):\n    \"\"\"\n    Scorer function for GridSearchCV that calculates -KL divergence\n    (negative because GridSearchCV maximizes score)\n    \"\"\"\n    y_pred = estimator.predict_proba(X)\n    return -kl_divergence(y, y_pred)  # Use soft labels directly\n\n# Load the training labels\ntrain_df = pd.read_csv(TRAIN_LABELS_PATH)\nprint(f\"Loaded {len(train_df)} training samples\")\n\n# Aggregate annotations by eeg_id\nprint(\"Aggregating annotations by eeg_id...\")\nvote_columns = [f\"{cls.lower()}_vote\" for cls in CLASSES]\naggregated_labels = []\n\ngrouped = train_df.groupby('eeg_id')\neeg_ids = []\ny_soft = []\n\nfor eeg_id, group in tqdm(grouped, total=len(grouped)):\n    votes = group[vote_columns].sum()\n    total_votes = votes.sum()\n    if total_votes == 0:\n        print(f\"Warning: No votes for eeg_id {eeg_id}, skipping...\")\n        continue\n    probs = votes / total_votes\n    eeg_ids.append(eeg_id)\n    y_soft.append(probs.values)\n\ny_soft = np.array(y_soft)\nprint(f\"Aggregated to {len(eeg_ids)} unique EEG IDs\")\n\n# Load features\nprint(\"Loading features...\")\nX = []\n\nsuccess_df = pd.read_csv(os.path.join(PREPROCESSED_PATH, \"success.csv\"))\nsuccess_ids = set(success_df['eeg_id'].tolist())\n\ndef load_features(eeg_id):\n    try:\n        eeg_path = os.path.join(PREPROCESSED_PATH, \"eeg\", f\"{eeg_id}.npy\")\n        eeg_data = np.load(eeg_path).astype(np.float32)\n        eeg_features = eeg_data.flatten()\n        \n        spec_path = os.path.join(PREPROCESSED_PATH, \"spec\", f\"{eeg_id}.npy\")\n        spec_data = np.load(spec_path).astype(np.float32)\n        spec_features = spec_data.flatten()\n        \n        features = np.concatenate([eeg_features, spec_features])\n        return features\n    except Exception as e:\n        print(f\"Error loading features for {eeg_id}: {e}\")\n        return None\n\nN = 17300\nindices_to_remove = []\n\nfor idx, eeg_id in tqdm(enumerate(eeg_ids), total=min(len(eeg_ids), N)):\n    if idx >= N:\n        break\n    if eeg_id not in success_ids:\n        indices_to_remove.append(idx)\n        continue\n    features = load_features(eeg_id)\n    if features is not None:\n        X.append(features)\n    else:\n        indices_to_remove.append(idx)\n\n# Remove entries with missing features\nfor idx in sorted(indices_to_remove, reverse=True):\n    y_soft = np.delete(y_soft, idx, axis=0)\n    eeg_ids.pop(idx)\n\nX = np.array(X)\nprint(f\"Loaded features for {len(X)} samples\")\nprint(f\"Feature vector shape: {X.shape}\")\nprint(f\"Label shape: {y_soft.shape}\")\n\nmax_cores = min(4, multiprocessing.cpu_count() - 1)\nprint(f\"Using {max_cores} cores for parallel processing\")\n\n# Hard labels for stratification\ny_hard = np.argmax(y_soft, axis=1)\n\nn_splits = 5\ncv = StratifiedKFold(n_splits=n_splits, shuffle=True, random_state=42)\n\nreduced_param_grid = {\n    'n_estimators': [100, 200],\n    'max_depth': [4, 8],\n    'learning_rate': [0.05, 0.1],\n    'subsample': [0.8, 1.0]\n}\n\nbase_xgb = XGBClassifier(\n    objective='multi:softprob',\n    num_class=len(CLASSES),\n    eval_metric='mlogloss',\n    use_label_encoder=False,\n    n_jobs=1,\n    verbosity=1,\n    random_state=42\n)\n\nprint(\"Training with cross-validation...\")\n\ncv_scores = []\ncv_models = []\ncv_predictions = []\ncv_feature_importances = []\nall_fold_predictions = {}\n\nfor fold, (train_idx, val_idx) in enumerate(cv.split(X, y_hard)):\n    print(f\"\\nFold {fold+1}/{n_splits}\")\n    X_train, X_val = X[train_idx], X[val_idx]\n    y_train, y_val = y_soft[train_idx], y_soft[val_idx]\n    eeg_ids_val = [eeg_ids[i] for i in val_idx]\n\n    grid_search = GridSearchCV(\n        estimator=base_xgb,\n        param_grid=reduced_param_grid,\n        scoring=kl_divergence_scorer,\n        cv=2,\n        n_jobs=1,\n        verbose=1\n    )\n    grid_search.fit(X_train, y_train.argmax(axis=1))\n\n    best_model = grid_search.best_estimator_\n    print(f\"Best parameters: {grid_search.best_params_}\")\n\n    y_val_pred_proba = best_model.predict_proba(X_val)\n    kl_score = kl_divergence(y_val, y_val_pred_proba)\n    print(f\"Fold {fold+1} KL Divergence: {kl_score:.6f}\")\n    ll_score = log_loss(y_val.argmax(axis=1), y_val_pred_proba)\n    print(f\"Fold {fold+1} Log Loss: {ll_score:.6f}\")\n\n    cv_scores.append(kl_score)\n    cv_models.append(best_model)\n    cv_predictions.append((y_val, y_val_pred_proba))\n    cv_feature_importances.append(best_model.feature_importances_)\n\n    for i, eeg_id in enumerate(eeg_ids_val):\n        all_fold_predictions[eeg_id] = y_val_pred_proba[i]\n\nos.makedirs(MODEL_OUTPUT_PATH, exist_ok=True)\nfor fold, model in enumerate(cv_models):\n    with open(os.path.join(MODEL_OUTPUT_PATH, f\"xgb_model_fold_{fold}.pkl\"), \"wb\") as f:\n        pickle.dump(model, f)\n\nmean_importance = np.mean(cv_feature_importances, axis=0)\nfeature_importance_df = pd.DataFrame({\n    'Feature': [f\"Feature_{i}\" for i in range(len(mean_importance))],\n    'Importance': mean_importance\n})\ntop_features = feature_importance_df.sort_values('Importance', ascending=False).head(30)\n\nplt.figure(figsize=(12, 8))\nsns.barplot(x='Importance', y='Feature', data=top_features)\nplt.title('Top 30 Feature Importances')\nplt.tight_layout()\nplt.savefig(os.path.join(MODEL_OUTPUT_PATH, 'feature_importance.png'))\nplt.close()\n\nmean_kl = np.mean(cv_scores)\nstd_kl = np.std(cv_scores)\nprint(f\"\\nCross-validation results:\")\nprint(f\"Mean KL Divergence: {mean_kl:.6f} ± {std_kl:.6f}\")\n\nprint(\"\\nTraining final model on all data with epoch-like output...\")\nfinal_model = XGBClassifier(\n    **grid_search.best_params_,\n    objective='multi:softprob',\n    num_class=len(CLASSES),\n    use_label_encoder=False,\n    eval_metric='mlogloss',\n    n_jobs=max_cores,\n    verbosity=1,\n    random_state=42\n)\nfinal_model.fit(X, y_soft.argmax(axis=1), eval_set=[(X, y_soft.argmax(axis=1))], verbose=True)\n\nwith open(os.path.join(MODEL_OUTPUT_PATH, \"xgb_model_final.pkl\"), \"wb\") as f:\n    pickle.dump(final_model, f)\n\nprint(\"Final model saved to:\", os.path.join(MODEL_OUTPUT_PATH, \"xgb_model_final.pkl\"))\n\nbest_fold = np.argmin(cv_scores)\ny_val, y_val_pred_proba = cv_predictions[best_fold]\ny_val_pred = np.argmax(y_val_pred_proba, axis=1)\ny_val_hard = np.argmax(y_val, axis=1)\n\nplt.figure(figsize=(10, 8))\ncm = confusion_matrix(y_val_hard, y_val_pred, normalize='true')\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=CLASSES)\ndisp.plot(cmap=plt.cm.Blues)\nplt.title(f'Normalized Confusion Matrix - Best Fold (KL: {cv_scores[best_fold]:.6f})')\nplt.tight_layout()\nplt.savefig(os.path.join(MODEL_OUTPUT_PATH, 'confusion_matrix.png'))\nplt.close()\n\ndef create_prediction_file():\n    test_eeg_path = os.path.join(BASE_PATH, \"test_eegs\")\n    test_files = [f.replace(\".parquet\", \"\") for f in os.listdir(test_eeg_path) if f.endswith(\".parquet\")]\n\n    submission_df = pd.DataFrame({'eeg_id': test_files})\n    for cls in CLASSES:\n        submission_df[cls] = 0.0\n\n    print(\"Generating predictions for test data...\")\n\n    for eeg_id in tqdm(test_files):\n        eeg_feature_path = os.path.join(\"/kaggle/working/preprocessed/eeg\", f\"{eeg_id}.npy\")\n        spec_feature_path = os.path.join(\"/kaggle/working/preprocessed/spec\", f\"{eeg_id}.npy\")\n\n        try:\n            eeg_features = np.load(eeg_feature_path)\n            spec_features = np.load(spec_feature_path)\n            features = np.concatenate([eeg_features.flatten(), spec_features.flatten()]).reshape(1, -1)\n        except Exception as e:\n            print(f\"Warning: Missing or error loading features for {eeg_id}. Error: {e}\")\n            features = None\n\n        if features is not None:\n            probs = final_model.predict_proba(features)[0]\n            for i, cls in enumerate(CLASSES):\n                submission_df.loc[submission_df['eeg_id'] == eeg_id, cls] = probs[i]\n        else:\n            uniform_prob = 1.0 / len(CLASSES)\n            for cls in CLASSES:\n                submission_df.loc[submission_df['eeg_id'] == eeg_id, cls] = uniform_prob\n\n    submission_path = \"/kaggle/working/submission.csv\"\n    submission_df.to_csv(submission_path, index=False)\n    print(\"Submission file saved to:\", submission_path)\n\n    return submission_df\n\nsubmission_df = create_prediction_file\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-26T17:09:11.324925Z","iopub.execute_input":"2025-05-26T17:09:11.325630Z","iopub.status.idle":"2025-05-26T17:09:23.667191Z","shell.execute_reply.started":"2025-05-26T17:09:11.325605Z","shell.execute_reply":"2025-05-26T17:09:23.666288Z"}},"outputs":[],"execution_count":null}]}