{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":4043,"databundleVersionId":44567,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# unzip train and test\n\nimport zipfile\nimport os\n\nzip_file_path = '/kaggle/input/inria-bci-challenge/train.zip'\nextract_to = '/kaggle/working/train'\n\nos.makedirs(extract_to, exist_ok=True)\n\nwith zipfile.ZipFile(zip_file_path, 'r') as zip_ref:\n    zip_ref.extractall(extract_to)\n\nprint(f\"File extracted to {extract_to}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:11:58.158466Z","iopub.execute_input":"2025-10-03T15:11:58.158831Z","iopub.status.idle":"2025-10-03T15:13:25.282510Z","shell.execute_reply.started":"2025-10-03T15:11:58.158811Z","shell.execute_reply":"2025-10-03T15:13:25.281777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"zip_file_path = '/kaggle/input/inria-bci-challenge/test.zip'\nextract_to = '/kaggle/working/test'\n\nos.makedirs(extract_to, exist_ok=True)\n\nwith zipfile.ZipFile(zip_file_path, 'r') as zip_ref:\n    zip_ref.extractall(extract_to)\n\nprint(f\"File extracted to {extract_to}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:14:24.777731Z","iopub.execute_input":"2025-10-03T15:14:24.778479Z","iopub.status.idle":"2025-10-03T15:15:20.207808Z","shell.execute_reply.started":"2025-10-03T15:14:24.778454Z","shell.execute_reply":"2025-10-03T15:15:20.207094Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\nimport glob\nimport mne\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom typing import Tuple, Optional, List","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:22:08.724888Z","iopub.execute_input":"2025-10-03T15:22:08.725470Z","iopub.status.idle":"2025-10-03T15:22:08.729269Z","shell.execute_reply.started":"2025-10-03T15:22:08.725447Z","shell.execute_reply":"2025-10-03T15:22:08.728660Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(os.listdir('/kaggle/working/train'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-01T11:00:33.422072Z","iopub.execute_input":"2025-10-01T11:00:33.422430Z","iopub.status.idle":"2025-10-01T11:00:33.427051Z","shell.execute_reply.started":"2025-10-01T11:00:33.422404Z","shell.execute_reply":"2025-10-01T11:00:33.426419Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/working/train/Data_S02_Sess01.csv')\nprint(df.columns)\ndf['FeedBackEvent'].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T14:49:17.758090Z","iopub.execute_input":"2025-10-02T14:49:17.758324Z","iopub.status.idle":"2025-10-02T14:49:18.777256Z","shell.execute_reply.started":"2025-10-02T14:49:17.758309Z","shell.execute_reply":"2025-10-02T14:49:18.776367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#shows the feedbacks which are equal to 1 using red dotted lines. the graph is of the\n#Cz channel.\nplt.figure(figsize=(14, 5))\nplt.plot(df['Time'], df['Pz'], label='Pz (brain channel)')\nfeedback_onsets = df[df['FeedBackEvent'] == 1]['Time']  \nfor t in feedback_onsets:\n    plt.axvline(x=t, color='r', linestyle='--', alpha=0.6)\nplt.xlabel('Time (ms or sample)')\nplt.ylabel('EEG amplitude')\nplt.title('EEG - Cz channel with feedback onsets')\nplt.legend()\nplt.show()\nnum_lines = len(feedback_onsets)\nprint(\"Number of red lines (feedback events):\", num_lines)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T14:49:49.777682Z","iopub.execute_input":"2025-10-02T14:49:49.778188Z","iopub.status.idle":"2025-10-02T14:49:50.654811Z","shell.execute_reply.started":"2025-10-02T14:49:49.778162Z","shell.execute_reply":"2025-10-02T14:49:50.654117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ndef create_mapping(output_csv):  \n    mapping_rows = []\n    \n    # Get list of EEG files\n    eeg_dir = '/kaggle/working/train'\n    eeg_files = [f for f in os.listdir(eeg_dir)]\n\n    preds_df = pd.read_csv('/kaggle/input/inria-bci-challenge/TrainLabels.csv')\n    pred_dict = dict(zip(preds_df[\"IdFeedBack\"], preds_df[\"Prediction\"]))\n\n    for eeg_filename in eeg_files:\n        eeg_path = os.path.join(eeg_dir, eeg_filename)\n        eeg = pd.read_csv(eeg_path)\n\n        # Remove Data word from the filename\n        subj_sess = os.path.splitext(eeg_filename)[0].replace('Data_', '')  # \"S02_sess01\"\n\n        # Extract rows where feedback is 1\n        fb_events = eeg[eeg['FeedBackEvent'] == 1].reset_index(drop=True)\n\n        # Label feedback values according to the time they occured \n        for idx, row in fb_events.iterrows():\n            event_num = idx + 1\n            fb_id = f\"{subj_sess}_FB{str(event_num).zfill(3)}\"\n            \n            mapping_rows.append({\n                'EEG_File': eeg_filename,\n                'Feedback_ID': fb_id,\n                'Time': row['Time'],\n                'Prediction': pred_dict.get(fb_id, None)\n            })\n    \n    mapping_df = pd.DataFrame(mapping_rows)\n    mapping_df.to_csv(output_csv, index=False)\n    \n    print(f\"Mapping created with {len(mapping_df)} rows. Saved to {output_csv}\")\n    return mapping_df\n    \nmapping_df = create_mapping(\n    output_csv=\"feedback_mapping.csv\"\n)\n\nprint(\"mapping_df.csv created successfully!\")\nprint(mapping_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:15:42.779099Z","iopub.execute_input":"2025-10-03T15:15:42.779488Z","iopub.status.idle":"2025-10-03T15:17:46.358449Z","shell.execute_reply.started":"2025-10-03T15:15:42.779464Z","shell.execute_reply":"2025-10-03T15:17:46.357569Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom scipy.signal import iirnotch, filtfilt\nfrom typing import List\n\ndef apply_notch_filter(df,eeg_channels: List[str], f0=50.0, Q=60.0, fs=200):\n    df_filtered = df.copy()\n    b, a = iirnotch(f0, Q, fs)\n    # Filter only the EEG channels, leaving other columns (like EOG) untouched\n    for channel in eeg_channels:\n        if channel in df_filtered.columns:\n            df_filtered[channel] = filtfilt(b, a, df_filtered[channel])\n            \n    return df_filtered\n\n# df = pd.read_csv('/kaggle/working/train/Data_S02_Sess01.csv')\n# filtered_eeg_channels_df = apply_notch_filter(df)\n# # Apply the function to your DataFrame\n# filtered_df = pd.concat([df[['Time', 'FeedBackEvent','EOG']], filtered_eeg_channels_df], axis=1)\n\n# plot_samples = 400 # Plot the first 2 seconds of data (400 samples / 200 Hz)\n# plt.figure(figsize=(15, 6))\n# plt.plot(df['Time'][:plot_samples], df['Cz'][:plot_samples], label='Original Cz Signal', color='red', alpha=1)\n# plt.plot(filtered_df['Time'][:plot_samples], filtered_df['Cz'][:plot_samples], label='Filtered Cz Signal (50Hz notch)', color='blue', linewidth=1)\n# plt.xlabel('Time (s)')\n# plt.ylabel('EEG Amplitude (µV)')\n# plt.title('Comparison of Original vs. Notch-Filtered EEG Signal (Cz Channel)')\n# plt.legend()\n# plt.grid(True, linestyle='--', alpha=0.6)\n# plt.show()\n\n# import matplotlib.pyplot as plt\n# from scipy.signal import welch\n\n# # Parameters\n# fs = 200  # Sampling frequency\n# nperseg = 1024  # segment length for Welch (controls smoothness)\n\n# # Get Cz channel signals\n# original_cz = df['Cz'].values\n# filtered_cz = filtered_df['Cz'].values\n\n# # Compute power spectral density (PSD) using Welch\n# f_orig, Pxx_orig = welch(original_cz, fs, nperseg=nperseg)\n# f_filt, Pxx_filt = welch(filtered_cz, fs, nperseg=nperseg)\n\n# # Plot frequency domain comparison\n# plt.figure(figsize=(15, 6))\n# plt.semilogy(f_orig, Pxx_orig, label='Original Cz Signal', color='red', alpha=0.8)\n# plt.semilogy(f_filt, Pxx_filt, label='Filtered Cz Signal (50Hz notch)', color='blue', alpha=0.8)\n\n# plt.axvline(50, color='k', linestyle='--', label='50 Hz (Notch)')\n# plt.xlabel('Frequency (Hz)')\n# plt.ylabel('Power Spectral Density (µV²/Hz)')\n# plt.title('Frequency Domain (Cz Channel) - Before vs After Notch Filtering')\n# plt.legend()\n# plt.grid(True, linestyle='--', alpha=0.6)\n# plt.xlim(0, 100)  # EEG is usually meaningful up to Nyquist (fs/2 = 100 Hz)\n# plt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:19:24.434811Z","iopub.execute_input":"2025-10-03T15:19:24.435250Z","iopub.status.idle":"2025-10-03T15:19:24.441097Z","shell.execute_reply.started":"2025-10-03T15:19:24.435229Z","shell.execute_reply":"2025-10-03T15:19:24.440351Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#reference: https://pmc.ncbi.nlm.nih.gov/articles/PMC6956025/\n#for removal of atifacts i've used a combination of linear regression model and ica \n# 1) linear regression is applied to remove linear artifacts\n# 2) seperating into independent components\n# 3) eog corelation from the given eog data\n# 4) wavelet enhancement, in the form of component correciton rather than component rejection\n# 5) reconstruction\nimport mne\nimport numpy as np\nimport pywt\nimport pandas as pd\nfrom scipy.signal import find_peaks # Import find_peaks\nimport matplotlib.pyplot as plt\n\ndef clean_eeg_pipeline_corrected(\n    df: pd.DataFrame,\n    sfreq: int ,\n    eeg_channels: list,\n    eog_channels: list,\n    l_freq: float = 0.5,\n    h_freq: float = 40.0,\n    n_components: int = 56,\n    random_state: int = 97,\n    wavelet: str = 'db4',\n    wavelet_level: int = 4,\n    wavelet_thresh_factor: float = 3.0,\n    peak_thresh_sd: float = 2.5, # Threshold in std devs for peak detection\n    peak_distance_s: float = 0.5 # Min distance between peaks in seconds\n) -> dict:\n    \"\"\"\n    peak_thresh_sd (float, optional): Standard deviation multiplier to set the\n                                     threshold for detecting artifact peaks.\n    peak_distance_s (float, optional): Minimum required distance between detected\n                                       peaks in seconds.\n    \"\"\"\n    # Steps 0-2\n    ch_names = eeg_channels + eog_channels\n    ch_types = ['eeg'] * len(eeg_channels) + ['eog'] * len(eog_channels)\n    info = mne.create_info(ch_names=ch_names, sfreq=sfreq, ch_types=ch_types)\n    raw = mne.io.RawArray(df[ch_names].T.values * 1e-6, info, verbose='ERROR')\n\n\n    raw_unfiltered = raw.copy()\n    raw.filter(l_freq, h_freq, fir_design='firwin', skip_by_annotation='edge')\n\n    eeg_data = raw.get_data(picks=eeg_channels)\n    eog_data = raw.get_data(picks=eog_channels)\n    eeg_cleaned_reg = np.zeros_like(eeg_data)\n\n    #apply linear regression\n    for i in range(eeg_data.shape[0]):\n        beta = np.linalg.lstsq(eog_data.T, eeg_data[i, :], rcond=None)[0]\n        eeg_cleaned_reg[i, :] = eeg_data[i, :] - eog_data.T @ beta\n    raw_reg_cleaned = raw.copy()\n    raw_reg_cleaned._data[:len(eeg_channels), :] = eeg_cleaned_reg\n\n    # Step 3: ICA Decomposition\n    ica = mne.preprocessing.ICA(n_components=n_components, random_state=random_state, max_iter='auto')\n    ica.fit(raw_reg_cleaned)\n\n    # Step 4: Identify EOG components to be CORRECTED\n    eog_indices_to_correct, eog_scores = ica.find_bads_eog(raw_reg_cleaned, ch_name=eog_channels)\n\n    # Step 5: Wavelet Component Correction \n    print(f\"Found {len(eog_indices_to_correct)} EOG-related components to correct: {eog_indices_to_correct}\")\n\n    # Get the original source signals\n    ica_sources = ica.get_sources(raw_reg_cleaned).get_data()\n    # Create a copy that we will modify\n    corrected_sources = ica_sources.copy()\n\n    def wavelet_denoise_segment(segment):\n        coeffs = pywt.wavedec(segment, wavelet, level=wavelet_level)\n        for i in range(1, len(coeffs)):\n            sigma = np.median(np.abs(coeffs[i])) / 0.6745\n            threshold = wavelet_thresh_factor * sigma\n            coeffs[i] = pywt.threshold(coeffs[i], threshold, mode='soft')\n        return pywt.waverec(coeffs, wavelet)\n\n    # Loop over only the components identified as EOG-related\n    for ic_index in eog_indices_to_correct:\n        ic_signal = ica_sources[ic_index, :]\n        \n        # 1. Detect peaks (blinks/saccades) in the component\n        peak_height_thresh = np.std(ic_signal) * peak_thresh_sd\n        min_peak_dist = int(sfreq * peak_distance_s)\n        peaks, _ = find_peaks(np.abs(ic_signal), height=peak_height_thresh, distance=min_peak_dist)\n        \n        if len(peaks) == 0:\n            continue # No artifacts found, move to the next component\n\n        # 2. Define 1-second windows around each peak\n        window_radius = sfreq // 2\n        artifact_windows = [(max(0, p - window_radius), min(len(ic_signal), p + window_radius)) for p in peaks]\n\n        # 3. Merge overlapping windows to create continuous artifact segments\n        if not artifact_windows:\n            continue\n        \n        # Sort windows by start time\n        artifact_windows.sort(key=lambda interval: interval[0])\n        merged = [artifact_windows[0]]\n        for current in artifact_windows[1:]:\n            prev = merged[-1]\n            if current[0] <= prev[1]: # Overlap\n                merged[-1] = (prev[0], max(prev[1], current[1]))\n            else:\n                merged.append(current)\n        \n        # 4. Apply wavelet denoising only within the merged artifact segments\n        for start, end in merged:\n            original_segment = ic_signal[start:end]\n            corrected_segment = wavelet_denoise_segment(original_segment)\n            corrected_segment = corrected_segment[:len(original_segment)]\n            # Place the corrected segment back into our sources array\n            corrected_sources[ic_index, start:end] = corrected_segment\n\n     # Step 6: Reconstruct Clean EEG\n    cleaned_eeg_data = np.dot(ica.mixing_matrix_[:, :ica.n_components_], corrected_sources)\n    \n    # Create a new Raw object for the cleaned data\n    info_eeg_only = mne.create_info(ch_names=eeg_channels, sfreq=sfreq, ch_types='eeg')\n    raw_after = mne.io.RawArray(cleaned_eeg_data, info_eeg_only)\n\n    return {\n        'cleaned_data': cleaned_eeg_data,\n        # 'raw_before': raw_reg_cleaned, # Data after filtering but before ICA correction\n        # 'raw_after': raw_after,\n        # 'ica': ica,\n        # 'eog_indices': eog_indices_to_correct\n    }\n\n# df = pd.read_csv('/kaggle/working/train/Data_S02_Sess01.csv')\n# eeg_channels = [col for col in df.columns if col not in ['Time', 'FeedBackEvent','EOG']]\n# eog_channels = ['EOG'] \n# print(f\"Data loaded successfully.\")\n# print(f\"Found {len(eeg_channels)} EEG channels.\")\n# sfreq=200\n# results = clean_eeg_pipeline_corrected(\n#     df=df,\n#     sfreq = sfreq,\n#     eeg_channels=eeg_channels,\n#     eog_channels=eog_channels,\n#     n_components=len(eeg_channels) \n# )\n\n# # --- Configuration for the plot ---\n# channel_to_plot_idx = 0  # Plot the first EEG channel\n# channel_name = eeg_channels[channel_to_plot_idx] # Get its name\n# duration_to_plot_s = 5  # seconds\n# start_time_s = 10      # Start plotting from 10 seconds into the data\n\n# # Calculate sample indices for the plotting duration\n# start_sample = int(start_time_s * sfreq)\n# end_sample = int((start_time_s + duration_to_plot_s) * sfreq)\n\n# # Get the data for the selected channel\n# # Ensure raw_before and raw_after objects exist in results\n# if 'raw_before' in results and 'raw_after' in results:\n#     # Extract data for the specific channel\n#     original_eeg = results['raw_before'].get_data(picks=channel_name)[0, start_sample:end_sample] * 1e6 # convert to uV\n#     cleaned_eeg = results['raw_after'].get_data(picks=channel_name)[0, start_sample:end_sample] * 1e6 # convert to uV\n\n#     # Create a time vector for plotting\n#     time = np.arange(0, len(original_eeg)) / sfreq\n\n#     # --- Plotting ---\n#     plt.figure(figsize=(12, 6))\n#     plt.plot(time, original_eeg, color='red', linewidth=0.7, label=f'Original {channel_name} Signal (Before Cleaning)')\n#     plt.plot(time, cleaned_eeg, color='blue', linewidth=0.7, label=f'Cleaned {channel_name} Signal (After Cleaning)')\n\n#     plt.title(f'Comparison of {channel_name} Channel Before vs. After EOG Cleaning')\n#     plt.xlabel('Time (s)')\n#     plt.ylabel('EEG Amplitude (µV)')\n#     plt.legend()\n#     plt.grid(True, linestyle='--', alpha=0.6)\n#     plt.tight_layout()\n#     plt.show()\n# else:\n#     print(\"Error: 'raw_before' or 'raw_after' not found in results. Please ensure the pipeline returns these.\")\n\n# print(\"Plotting the power spectrum before and after cleaning...\")\n# fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(15, 5), sharey=True)\n# results['raw_before'].plot_psd(ax=ax1, show=False, fmax=50)\n# ax1.set_title('PSD Before Cleaning')\n# ax1.set_xlabel('Frequency (Hz)')\n# results['raw_after'].plot_psd(ax=ax2, show=False, fmax=50)\n# ax2.set_title('PSD After Cleaning')\n# ax2.set_xlabel('Frequency (Hz)')\n# fig.suptitle('Power Spectral Density Comparison')\n# plt.tight_layout()\n# plt.show()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:19:43.247668Z","iopub.execute_input":"2025-10-03T15:19:43.248352Z","iopub.status.idle":"2025-10-03T15:19:43.262744Z","shell.execute_reply.started":"2025-10-03T15:19:43.248329Z","shell.execute_reply":"2025-10-03T15:19:43.262173Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_epochs_with_rejection(\n    eeg_csv_path: str, feedback_df_group: pd.DataFrame, sfreq: float,\n    all_eeg_chans: List[str], feature_chans: List[str],\n    tmin: float, tmax: float, reject_threshold_uv: float = 150.0\n) -> Optional[mne.Epochs]:\n    \"\"\"\n    A simpler pipeline that loads data, applies standard filters, and uses\n    peak-to-peak amplitude rejection to remove major artifacts.\n    \"\"\"\n    print(f\"[INFO] Reading continuous data from: {os.path.basename(eeg_csv_path)}\")\n    try:\n        eeg_df = pd.read_csv(eeg_csv_path, usecols=all_eeg_chans)\n    except (FileNotFoundError, ValueError) as e:\n        print(f\"[ERROR] Could not read or find required channels in {eeg_csv_path}: {e}\")\n        return None\n\n    # 1. Create MNE Raw object, assuming input is in µV and converting to V\n    info = mne.create_info(ch_names=all_eeg_chans, sfreq=sfreq, ch_types='eeg')\n    raw = mne.io.RawArray(eeg_df[all_eeg_chans].T.values * 1e-6, info, verbose='ERROR')\n    \n    # 2. Apply a standard band-pass filter to remove slow drifts and high-frequency noise\n    raw.filter(l_freq=1.0, h_freq=40.0, fir_design='firwin', verbose='ERROR')\n    \n    # 3. Select only the channels we need for feature extraction\n    raw.pick_channels(feature_chans)\n    \n    # 4. Create an MNE events array from feedback timestamps\n    event_samples = (feedback_df_group['Time'].values * sfreq).astype(int)\n    events = np.vstack([event_samples, np.zeros_like(event_samples), np.ones_like(event_samples)]).T\n\n    # 5. Create the Epochs object WITH ARTIFACT REJECTION\n    # This is the key step. It will automatically drop any epoch where the\n    # signal jumps by more than the rejection threshold.\n    reject_criteria = dict(eeg=reject_threshold_uv * 1e-6) # Convert µV to V for MNE\n\n    try:\n        epochs = mne.Epochs(\n            raw, events=events, tmin=tmin, tmax=tmax,\n            baseline=(None, 0),\n            reject=reject_criteria, # <-- THE NEW PARAMETER\n            preload=True,\n            verbose='ERROR'\n        )\n        \n        # We need to know which events were kept to align metadata\n        n_events_original = len(events)\n        n_epochs_created = len(epochs)\n        if n_events_original != n_epochs_created:\n            print(f\"[INFO] Dropped {n_events_original - n_epochs_created} epochs due to artifacts or boundary issues.\")\n\n        metadata_kept = feedback_df_group.iloc[epochs.selection].reset_index(drop=True)\n        epochs.metadata = metadata_kept\n        \n        return epochs\n    except ValueError as e:\n        print(f\"[WARN] Could not create epochs for {os.path.basename(eeg_csv_path)}. Error: {e}\")\n        return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:22:15.526499Z","iopub.execute_input":"2025-10-03T15:22:15.527186Z","iopub.status.idle":"2025-10-03T15:22:16.316328Z","shell.execute_reply.started":"2025-10-03T15:22:15.527161Z","shell.execute_reply":"2025-10-03T15:22:16.315769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# EEG feature extraction \n# https://pmc.ncbi.nlm.nih.gov/articles/PMC1868547/\n# the relevant extraction is done using ERP (Event-Related Potentials)\n# 1. average the signal in short time windows after feedback\n# (e.g., 200–300 ms for FRN (Feedback Related Negativity) at FCz/Cz, 300–600 ms for P300 (a positive peak around 300 ms after an infrequent target appears) at Pz).\n# 2. compute 4–8 Hz power ~200–400 ms at FCz \n# the brain also produces short burst of specific rhythm (a \"theta wave\") when processing feedback.\n\nfrom __future__ import annotations\nimport os\nfrom typing import Tuple, Optional, List\n\nimport numpy as np\nimport pandas as pd\nimport mne\nfrom scipy.signal import hilbert, iirnotch, filtfilt, find_peaks\nimport pywt # You may need to run: pip install PyWavelets\n\n# ==============================================================================\n# CONFIG — EDIT THESE VALUES\n# ==============================================================================\n# --- Input Paths ---\nFEEDBACK_CSV_PATH = \"/kaggle/working/feedback_mapping.csv\"\nEEG_DATA_DIR      = \"/kaggle/working/train\"\n\n# --- File & Column Naming ---\nEEG_FILE_EXTENSION = \".csv\"\nFILE_COL           = \"EEG_File\"\nTIMESTAMP_COL      = \"Time\"\nLABEL_COL          = \"prediction\"\n# **NEW**: Define both EEG and EOG channel names from your CSV files\nEEG_CHANNEL_COLS   = ['FCz', 'Cz', 'Pz'] # Channels to use for feature extractionA\nEOG_CHANNEL_COLS   = ['EOG'] # The name of your EOG channel column\n# OPTIONAL_META      = [\"subject_id\", \"dataset_id\", \"session_id\", \"trial_id\"]\n\n# --- Preprocessing & Epoching Parameters ---\nSAMPLING_RATE_HZ = 200.0  # The sampling frequency of your EEG data\nEPOCH_TMIN       = -0.2   # Start time of an epoch in seconds\nEPOCH_TMAX       = 1.0    # End time of an epoch in seconds\n\n# --- Output Files ---\nOUTPUT_CSV      = \"all_training_data_cleaned.csv\"\nOUTPUT_PICKLE   = \"all_training_data_cleaned.pkl\"\n\n# ==============================================================================\n# UPDATED HELPER: Create Epochs from Cleaned CSV Data\n# ==============================================================================\ndef create_epochs_from_csv(\n    eeg_csv_path: str, feedback_df_group: pd.DataFrame, sfreq: float,\n    all_eeg_chans: List[str], eog_chans: List[str], feature_chans: List[str],\n    tmin: float, tmax: float\n) -> Optional[mne.Epochs]:\n    \"\"\"\n    Loads, cleans (notch + ICA), and epochs the EEG data from a CSV file.\n    \"\"\"\n    print(f\"[INFO] Reading continuous data from: {os.path.basename(eeg_csv_path)}\")\n    try:\n        # 1. Load the raw data from the CSV file\n        eeg_df = pd.read_csv(eeg_csv_path, usecols=all_eeg_chans + eog_chans)\n    except (FileNotFoundError, ValueError) as e:\n        print(f\"[ERROR] Could not read or find required channels in {eeg_csv_path}: {e}\")\n        return None\n    \n    # 2. Apply the 50Hz notch filter\n    print(\"Applying 50Hz notch filter...\")\n    eeg_df_notched = apply_notch_filter(eeg_df, eeg_channels=all_eeg_chans, fs=sfreq)\n\n    # 3. Apply the advanced cleaning pipeline (ICA + Wavelet)\n    print(\"Applying advanced EOG artifact correction pipeline...\")\n    cleaning_results = clean_eeg_pipeline_corrected(\n        df=eeg_df_notched, sfreq=int(sfreq), eeg_channels=all_eeg_chans, eog_channels=eog_chans\n    )\n    cleaned_data = cleaning_results['cleaned_data'] # This data is now in Volts\n\n    # 4. Create an MNE Info object and RawArray from the CLEANED data\n    info = mne.create_info(ch_names=all_eeg_chans, sfreq=sfreq, ch_types='eeg')\n    raw_cleaned = mne.io.RawArray(cleaned_data, info, verbose='ERROR')\n    \n    # 5. Select only the channels needed for feature extraction\n    raw_cleaned.pick_channels(feature_chans)\n\n    # 6. Create an MNE events array from feedback timestamps\n    event_samples = (feedback_df_group[TIMESTAMP_COL].values * sfreq).astype(int)\n    events = np.vstack([event_samples, np.zeros_like(event_samples), np.ones_like(event_samples)]).T\n\n    # 7. Create the Epochs object\n    try:\n        epochs = mne.Epochs(\n            raw_cleaned, events=events, tmin=tmin, tmax=tmax,\n            baseline=(None, 0), preload=True, verbose='ERROR'\n        )\n    except ValueError as e:\n        print(f\"[WARN] Could not create epochs for {os.path.basename(eeg_csv_path)}. Error: {e}\")\n        return None\n\n    epochs.metadata = feedback_df_group.copy().reset_index(drop=True)\n    return epochs\n\ndef _time_to_samples(epochs: mne.Epochs, tmin_ms: float, tmax_ms: float) -> Tuple[int, int]:\n    t = epochs.times\n    i0 = np.searchsorted(t, tmin_ms / 1000.0, side='left')\n    i1 = np.searchsorted(t, tmax_ms / 1000.0, side='right')\n    return i0, i1\n\ndef _theta_power_hilbert(epochs: mne.Epochs, ch_name: str, tmin_ms: float, tmax_ms: float, fmin: float = 4., fmax: float = 8.) -> np.ndarray:\n    ep_ch = epochs.copy().pick(ch_name).filter(fmin, fmax, method='fir', phase='zero', verbose='ERROR')\n    data = ep_ch.get_data()[:, 0, :]\n    analytic = hilbert(data, axis=-1)\n    power = np.abs(analytic) ** 2\n    i0, i1 = _time_to_samples(ep_ch, tmin_ms, tmax_ms)\n    return power[:, i0:i1].mean(axis=1) if i1 > i0 else np.full((power.shape[0],), np.nan)\n\ndef extract_bci_features(epochs: mne.Epochs, to_microvolts: bool = True) -> pd.DataFrame:\n    frn_win, p300_win, theta_win = (200, 300), (300, 600), (200, 400)\n    missing = [ch for ch in EEG_CHANNEL_COLS if ch not in epochs.ch_names]\n    if missing: raise ValueError(f\"Missing required feature channels: {missing}.\")\n\n    X = epochs.get_data()\n    if to_microvolts: X *= 1e6\n\n    i0_frn, i1_frn = _time_to_samples(epochs, *frn_win)\n    i0_p3, i1_p3 = _time_to_samples(epochs, *p300_win)\n\n    idx_FCz, idx_Cz, idx_Pz = [epochs.ch_names.index(ch) for ch in EEG_CHANNEL_COLS]\n\n    frn_FCz = X[:, idx_FCz, i0_frn:i1_frn].mean(axis=1)\n    frn_Cz = X[:, idx_Cz, i0_frn:i1_frn].mean(axis=1)\n    p300_Pz = X[:, idx_Pz, i0_p3:i1_p3].mean(axis=1)\n    theta_power_FCz = _theta_power_hilbert(epochs, 'FCz', *theta_win)\n\n    return pd.DataFrame({\n        'FRN_FCz_mean_200_300ms': frn_FCz,\n        'FRN_Cz_mean_200_300ms': frn_Cz,\n        'P300_Pz_mean_300_600ms': p300_Pz,\n        'Theta_4_8Hz_power_FCz_200_400ms': theta_power_FCz,\n    })\n\n# ==============================================================================\n# Main Build Function (Updated to call new epoch creation function)\n# ==============================================================================\ndef build_training_table() -> pd.DataFrame:\n    fb = pd.read_csv(FEEDBACK_CSV_PATH)\n    fb[\"_file_key\"] = fb[FILE_COL].str.replace(r'\\.csv$', '', regex=True)\n    all_rows = []\n    for file_key, group in fb.groupby(\"_file_key\"):\n        print(f\"\\n---> Now processing file: {file_key}, which has {len(group)} events listed in the CSV.\")\n        # if file_key == 'Data_S02_Sess01':\n        #     print(\"\\n--- RUNNING DIAGNOSTICS for data_s02_sess01 ---\")\n            \n        #     # 1. Get all timestamps for this file (in seconds)\n        #     timestamps_s = group[TIMESTAMP_COL].values\n            \n        #     # 2. Check for NaN or missing timestamps\n        #     nan_count = np.isnan(timestamps_s).sum()\n        #     if nan_count > 0:\n        #         print(f\"[PROBLEM] Found {nan_count} NaN timestamps!\")\n    \n        #     # 3. Define boundaries\n        #     start_boundary = -EPOCH_TMIN  # This will be 0.2 seconds\n        #     # To get the end boundary, we have to load the EEG data first\n        #     eeg_csv_path = os.path.join(EEG_DATA_DIR, f\"{file_key}{EEG_FILE_EXTENSION}\")\n        #     try:\n        #         # We only need to load it to check the duration\n        #         eeg_df = pd.read_csv(eeg_csv_path)\n        #         eeg_duration_s = len(eeg_df) / SAMPLING_RATE_HZ\n        #         end_boundary = eeg_duration_s - EPOCH_TMAX\n        #         print(f\"Verified EEG recording duration: {eeg_duration_s:.3f} seconds\")\n        #     except FileNotFoundError:\n        #         print(\"[ERROR] Could not find the EEG file to verify duration.\")\n        #         continue # Skip to next file in the main loop\n    \n        #     # 4. Find and list the events that violate the boundaries\n        #     early_events = timestamps_s[timestamps_s < start_boundary]\n        #     late_events = timestamps_s[timestamps_s > end_boundary]\n    \n        #     print(f\"\\nChecking against start boundary ({start_boundary:.3f}s) and end boundary ({end_boundary:.3f}s)...\")\n    \n        #     if len(early_events) > 0:\n        #         print(f\"[PROBLEM] Found {len(early_events)} events that are too close to the start:\")\n        #         print(np.round(early_events, 3))\n            \n        #     if len(late_events) > 0:\n        #         print(f\"[PROBLEM] Found {len(late_events)} events that are too close to the end:\")\n        #         print(np.round(late_events, 3))\n                \n        #     if len(early_events) == 0 and len(late_events) == 0 and nan_count == 0:\n        #         print(\"[INFO] No boundary violations found for this file.\")\n    \n        #     print(\"--- END DIAGNOSTICS ---\\n\")\n            \n        eeg_csv_path = os.path.join(EEG_DATA_DIR, f\"{file_key}{EEG_FILE_EXTENSION}\")\n        if not os.path.exists(eeg_csv_path):\n            print(f\"[WARN] Could not find EEG CSV for '{file_key}'. Skipping.\")\n            continue\n        else:\n            df = pd.read_csv(eeg_csv_path)\n        ALL_EEG_CHANNELS = [col for col in df.columns if col not in ['Time', 'FeedBackEvent', 'EOG']]\n    \n        # Call the updated function that now includes the cleaning steps\n        epochs = create_epochs_with_rejection(\n            eeg_csv_path, group, sfreq=SAMPLING_RATE_HZ,\n            all_eeg_chans=ALL_EEG_CHANNELS,\n            feature_chans=EEG_CHANNEL_COLS, tmin=EPOCH_TMIN, tmax=EPOCH_TMAX\n        )\n\n        if epochs is None or len(epochs) == 0:\n            print(f\"[WARN] No valid epochs were created for '{file_key}'. Skipping.\")\n            continue\n\n        feats = extract_bci_features(epochs)\n        out = epochs.metadata.merge(feats, left_index=True, right_index=True)\n        all_rows.append(out)\n\n    if not all_rows:\n        raise RuntimeError(\"No data could be processed. Check file paths and channel names.\")\n\n    all_data = pd.concat(all_rows, axis=0, ignore_index=True)\n    \n    feature_cols = [c for c in all_data.columns if c.startswith((\"FRN_\", \"P300_\", \"Theta_\"))]\n    all_data.dropna(subset=feature_cols, inplace=True)\n    all_data.reset_index(drop=True, inplace=True)\n\n    all_data.to_csv(OUTPUT_CSV, index=False)\n    print(f\"\\n[SUCCESS] Saved unified training table to: {OUTPUT_CSV} ({len(all_data)} rows)\")\n    all_data.to_pickle(OUTPUT_PICKLE)\n    return all_data\n\n# ==============================================================================\n# Run the script\n# ==============================================================================\nif __name__ == \"__main__\":\n    mne.set_log_level('WARNING')\n    final_df = build_training_table()\n    print(\"\\n--- First 5 rows of the final cleaned dataset ---\")\n    print(final_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:22:20.130568Z","iopub.execute_input":"2025-10-03T15:22:20.130814Z","iopub.status.idle":"2025-10-03T15:26:35.306397Z","shell.execute_reply.started":"2025-10-03T15:22:20.130797Z","shell.execute_reply":"2025-10-03T15:26:35.305610Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Load the dataset you created\ndf = pd.read_csv(\"all_training_data_cleaned.csv\")\n\n# Pick a key feature, like the P300\nfeature_to_plot = 'P300_Pz_mean_300_600ms'\ntarget_col = 'Prediction' # Or 'label' if you named it that\n\nplt.figure(figsize=(10, 6))\nsns.kdeplot(data=df, x=feature_to_plot, hue=target_col, fill=True, common_norm=False)\nplt.title(f'Distribution of {feature_to_plot} by Class')\nplt.xlabel('Mean Amplitude (µV)')\nplt.grid(True, linestyle='--')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:28:36.642312Z","iopub.execute_input":"2025-10-03T15:28:36.643205Z","iopub.status.idle":"2025-10-03T15:28:37.716135Z","shell.execute_reply.started":"2025-10-03T15:28:36.643174Z","shell.execute_reply":"2025-10-03T15:28:37.715284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport mne\nimport matplotlib.pyplot as plt\n\n# ==============================================================================\n# IMPORTANT: PASTE ALL YOUR HELPER FUNCTIONS HERE\n# You need the full definitions of:\n# - apply_notch_filter\n# - clean_eeg_pipeline_corrected\n# - create_epochs_from_csv\n# - _time_to_samples\n# - _theta_power_hilbert\n# - extract_bci_features\n# ==============================================================================\n# (All helper functions from your main script would go here)\n\n\n# ==============================================================================\n# CONFIGURATION FOR VISUALIZATION\n# ==============================================================================\n# --- Choose the single file you want to analyze ---\nFILE_TO_ANALYZE = \"Data_S02_Sess01\" # The file that had issues, for example\n\n# --- Set paths and parameters (should match your main script) ---\nFEEDBACK_CSV_PATH = \"/kaggle/working/feedback_mapping.csv\"\nEEG_DATA_DIR      = \"/kaggle/working/train\"\nEEG_FILE_EXTENSION = \".csv\"\nFILE_COL           = \"EEG_File\"\nTIMESTAMP_COL      = \"Time\"\nEEG_CHANNEL_COLS   = ['FCz', 'Cz', 'Pz']\nEOG_CHANNEL_COLS   = ['EOG']\nSAMPLING_RATE_HZ = 200.0\nEPOCH_TMIN       = -0.2\nEPOCH_TMAX       = 1.0\n\n\n# ==============================================================================\n# SCRIPT TO GENERATE AND PLOT THE ERPs\n# ==============================================================================\n\n# 1. LOAD THE MAPPING FILE AND GET DATA FOR OUR CHOSEN FILE\nprint(\"Loading feedback mapping file...\")\nfb = pd.read_csv(FEEDBACK_CSV_PATH)\nfb[\"_file_key\"] = fb[FILE_COL].str.replace(r'\\.csv$', '', regex=True)\ngroup_to_process = fb[fb[\"_file_key\"] == FILE_TO_ANALYZE]\n\nif group_to_process.empty:\n    raise ValueError(f\"Could not find file key '{FILE_TO_ANALYZE}' in the feedback CSV.\")\n\n# 2. RUN THE FULL PROCESSING PIPELINE TO CREATE THE EPOCHS OBJECT\nprint(f\"\\nProcessing {FILE_TO_ANALYZE} to create epochs...\")\neeg_csv_path = os.path.join(EEG_DATA_DIR, f\"{FILE_TO_ANALYZE}{EEG_FILE_EXTENSION}\")\n\n# Dynamically get all channel names from the file\ndf_cols = pd.read_csv(eeg_csv_path, nrows=1).columns\nALL_EEG_CHANNELS = [col for col in df_cols if col not in ['Time', 'FeedBackEvent', 'EOG']]\n\n# This is the crucial step that was missing before\nepochs = create_epochs_with_rejection(\n    eeg_csv_path, group_to_process, sfreq=SAMPLING_RATE_HZ,\n    all_eeg_chans=ALL_EEG_CHANNELS,\n    feature_chans=EEG_CHANNEL_COLS, tmin=EPOCH_TMIN, tmax=EPOCH_TMAX\n)\n\nif epochs is None:\n    raise RuntimeError(\"Epoch creation failed for the selected file.\")\n\nprint(f\"Successfully created {len(epochs)} epochs.\")\n\n# 3. SEPARATE THE CREATED EPOCHS BY CLASS (0 vs 1)\n# Note: Use the correct column name from your metadata, e.g., 'prediction'\nepochs_class_0 = epochs[epochs.metadata['Prediction'] == 0]\nepochs_class_1 = epochs[epochs.metadata['Prediction'] == 1]\nprint(f\"Found {len(epochs_class_0)} epochs for class 0.\")\nprint(f\"Found {len(epochs_class_1)} epochs for class 1.\")\n\n\n# 4. AVERAGE THE EPOCHS FOR EACH CLASS TO CREATE \"EVOKED\" RESPONSES\nevoked_0 = epochs_class_0.average()\nevoked_1 = epochs_class_1.average()\n\n# 5. PLOT THE COMPARISON\nprint(\"\\nPlotting the average ERPs for each class...\")\n# We can plot multiple channels at once\nchannels_to_plot = ['FCz', 'Cz', 'Pz']\n\nfig = mne.viz.plot_compare_evokeds(\n    {'Class 0': evoked_0, 'Class 1': evoked_1},\n    picks=channels_to_plot,\n    title=f'Average ERP for Class 0 vs. Class 1 on file {FILE_TO_ANALYZE}'\n)\n# Make the plot bigger for better visibility\nfor figure in fig:\n    figure.set_size_inches(12, 8)\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:35:16.782069Z","iopub.execute_input":"2025-10-03T15:35:16.782346Z","iopub.status.idle":"2025-10-03T15:35:18.303617Z","shell.execute_reply.started":"2025-10-03T15:35:16.782324Z","shell.execute_reply":"2025-10-03T15:35:18.302950Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\n\n# Load one of your raw EEG files\ndf = pd.read_csv('/kaggle/working/train/Data_S02_Sess01.csv')\n\n# Check the statistics of a key channel\nprint(df['Pz'].describe())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:35:21.455087Z","iopub.execute_input":"2025-10-03T15:35:21.455671Z","iopub.status.idle":"2025-10-03T15:35:22.469955Z","shell.execute_reply.started":"2025-10-03T15:35:21.455647Z","shell.execute_reply":"2025-10-03T15:35:22.469244Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport joblib\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import accuracy_score, classification_report, confusion_matrix, roc_auc_score\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# --- NEW: Import LDA and SVM ---\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.svm import SVC\n\n# --- CONFIGURATION ---\nTRAINING_DATA_PATH = \"all_training_data_cleaned.csv\"\nMODEL_SAVE_PATH = \"eeg_classifier_model.joblib\"\nSCALER_SAVE_PATH = \"feature_scaler.joblib\"\nTARGET_COL = \"Prediction\"\n\n# 1. LOAD THE DATASET\nprint(f\"Loading feature-engineered dataset from: {TRAINING_DATA_PATH}\")\ndf = pd.read_csv(TRAINING_DATA_PATH)\n\n# 2. DEFINE FEATURES (X) AND TARGET (y)\nfeature_cols = [col for col in df.columns if col.startswith(('FRN_', 'P300_', 'Theta_'))]\nX = df[feature_cols]\ny = df[TARGET_COL]\n\nprint(f\"\\nFound {len(feature_cols)} features: {feature_cols}\")\n\n# 3. SPLIT DATA INTO TRAINING AND VALIDATION SETS\nX_train, X_val, y_train, y_val = train_test_split(\n    X, y, test_size=0.2, random_state=42, stratify=y\n)\nprint(f\"\\nTraining set: {len(X_train)} samples\")\nprint(f\"Validation set: {len(X_val)} samples\")\n\n# 4. SCALE THE FEATURES\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_val_scaled = scaler.transform(X_val)\n\n# 5. INITIALIZE AND TRAIN THE MODEL\nprint(\"\\n--- Model Selection ---\")\n\n# --- Option A: Linear Discriminant Analysis (LDA) ---\n# Fast, simple, and a very strong baseline for BCI.\nprint(\"Initializing LDA model...\")\nmodel = LinearDiscriminantAnalysis(solver='lsqr', shrinkage='auto')\n\n# --- Option B: Support Vector Machine (SVM) ---\n# Powerful but can be much slower to train, especially with large datasets.\n# To use SVM, comment out the LDA line above and uncomment the two lines below.\n# print(\"Initializing SVM model...\")\n# model = SVC(kernel='rbf', probability=True, random_state=42) # probability=True is essential!\n\n\nprint(f\"\\nTraining the {type(model).__name__} model...\")\nmodel.fit(X_train_scaled, y_train)\nprint(\"Training complete.\")\n\n# 6. SAVE THE TRAINED MODEL AND SCALER\njoblib.dump(model, MODEL_SAVE_PATH)\njoblib.dump(scaler, SCALER_SAVE_PATH)\nprint(f\"\\nModel saved to: {MODEL_SAVE_PATH}\")\nprint(f\"Scaler saved to: {SCALER_SAVE_PATH}\")\n\n# 7. EVALUATE ON THE VALIDATION SET (using probabilities)\nprint(\"\\n--- Evaluating on Validation Set ---\")\n# Get the probability predictions for the positive class (class 1)\ny_proba_val = model.predict_proba(X_val_scaled)[:, 1]\n\n# Calculate the primary metric: AUC score\nauc_score = roc_auc_score(y_val, y_proba_val)\nprint(f\"Validation AUC Score: {auc_score:.4f}\")\n\n# For a more traditional view, calculate accuracy at a 0.5 threshold\ny_pred_val_thresholded = (y_proba_val >= 0.5).astype(int)\naccuracy = accuracy_score(y_val, y_pred_val_thresholded)\nprint(f\"\\nValidation Accuracy (at 0.5 threshold): {accuracy:.4f}\")\nprint(\"\\nClassification Report (at 0.5 threshold):\")\nprint(classification_report(y_val, y_pred_val_thresholded))\n\n# Display the confusion matrix\ncm = confusion_matrix(y_val, y_pred_val_thresholded)\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues')\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.title('Validation Confusion Matrix (at 0.5 threshold)')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:35:24.707588Z","iopub.execute_input":"2025-10-03T15:35:24.707854Z","iopub.status.idle":"2025-10-03T15:35:25.767203Z","shell.execute_reply.started":"2025-10-03T15:35:24.707831Z","shell.execute_reply":"2025-10-03T15:35:25.766464Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ndef create_test_mapping(eeg_dir, output_csv):\n    \"\"\"\n    Scans a directory of raw EEG CSVs (test set), finds all feedback event\n    timestamps, and creates a mapping file (event manifest) without labels.\n    \"\"\"\n    mapping_rows = []\n    \n    # Get list of EEG files from the specified test directory\n    eeg_files = [f for f in os.listdir(eeg_dir) if f.endswith('.csv')]\n    print(f\"Found {len(eeg_files)} EEG files in {eeg_dir}\")\n\n    for eeg_filename in eeg_files:\n        eeg_path = os.path.join(eeg_dir, eeg_filename)\n        eeg = pd.read_csv(eeg_path)\n\n        # Remove Data word from the filename\n        subj_sess = os.path.splitext(eeg_filename)[0].replace('Data_', '') # e.g., \"S21_Sess01\"\n\n        # Extract rows where feedback is 1\n        fb_events = eeg[eeg['FeedBackEvent'] == 1].reset_index(drop=True)\n\n        # Create a unique ID for each feedback event for tracking\n        for idx, row in fb_events.iterrows():\n            event_num = idx + 1\n            # This ID is still useful for your final submission file\n            fb_id = f\"{subj_sess}_FB{str(event_num).zfill(3)}\"\n            \n            mapping_rows.append({\n                'EEG_File': eeg_filename,\n                'Feedback_ID': fb_id,\n                'Time': row['Time'],\n            })\n    \n    mapping_df = pd.DataFrame(mapping_rows)\n    mapping_df.to_csv(output_csv, index=False)\n    \n    print(f\"Test mapping created with {len(mapping_df)} rows. Saved to {output_csv}\")\n    return mapping_df\n\nTEST_EEG_DIR = '/kaggle/working/test'\ntest_mapping_df = create_test_mapping(\n    eeg_dir=TEST_EEG_DIR,\n    output_csv=\"test_feedback_mapping.csv\"\n)\n\nprint(\"\\nTest mapping created successfully!\")\nprint(test_mapping_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:35:34.277045Z","iopub.execute_input":"2025-10-03T15:35:34.277626Z","iopub.status.idle":"2025-10-03T15:36:43.980198Z","shell.execute_reply.started":"2025-10-03T15:35:34.277605Z","shell.execute_reply":"2025-10-03T15:36:43.979310Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from __future__ import annotations\nimport os\nfrom typing import Tuple, Optional, List\n\nimport numpy as np\nimport pandas as pd\nimport mne\nfrom scipy.signal import hilbert, iirnotch, filtfilt, find_peaks\nimport pywt # You may need to run: pip install PyWavelets\nimport joblib\n\n# ==============================================================================\n# CONFIG — VERIFY THESE VALUES MATCH YOUR TRAINING SETUP\n# ==============================================================================\n# --- Input Paths ---\nTEST_EEG_DIR = '/kaggle/working/test'\nTEST_FEEDBACK_MAPPING_CSV = \"test_feedback_mapping.csv\" # The event manifest you just created\nMODEL_PATH = \"eeg_classifier_model.joblib\" # Saved model from your training script\nSCALER_PATH = \"feature_scaler.joblib\"     # Saved scaler from your training script\n\n# --- File & Column Naming ---\nEEG_FILE_EXTENSION = \".csv\"\nFILE_COL           = \"EEG_File\"       # Column in your mapping CSV that identifies the EEG file\nTIMESTAMP_COL      = \"Time\"           # Column in your mapping CSV for timestamps (in seconds)\n# NOTE: The feature extraction channels MUST match what the model was trained on\nEEG_CHANNEL_COLS   = ['FCz', 'Cz', 'Pz']\nEOG_CHANNEL_COLS   = ['EOG']\n\n# --- Preprocessing & Epoching Parameters (MUST MATCH TRAINING) ---\nSAMPLING_RATE_HZ = 200.0\nEPOCH_TMIN       = -0.2\nEPOCH_TMAX       = 1.0\n\n# --- Output Files ---\nOUTPUT_PREDICTIONS_CSV = \"test_predictions.csv\"\nOUTPUT_SUBMISSION_CSV = \"submission.csv\"\n    \ndef process_single_file_to_features(file_key: str, group: pd.DataFrame, eeg_dir: str) -> Optional[pd.DataFrame]:\n    eeg_csv_path = os.path.join(eeg_dir, f\"{file_key}.csv\")\n    if not os.path.exists(eeg_csv_path):\n        print(f\"[WARN] Could not find EEG CSV for '{file_key}'. Skipping.\")\n        return None\n    try:\n        df_cols = pd.read_csv(eeg_csv_path, nrows=1).columns\n        ALL_EEG_CHANNELS = [col for col in df_cols if col not in ['Time', 'FeedBackEvent', 'EOG']]\n    except Exception as e:\n        print(f\"[ERROR] Could not read columns from {eeg_csv_path}: {e}\")\n        return None\n    epochs = create_epochs_from_csv(eeg_csv_path, group, sfreq=SAMPLING_RATE_HZ, all_eeg_chans=ALL_EEG_CHANNELS, eog_chans=EOG_CHANNEL_COLS, feature_chans=EEG_CHANNEL_COLS, tmin=EPOCH_TMIN, tmax=EPOCH_TMAX)\n    if epochs is None or len(epochs) == 0:\n        print(f\"[WARN] No valid epochs were created for '{file_key}'. Skipping.\")\n        return None\n    feats = extract_bci_features(epochs)\n    processed_df = epochs.metadata.merge(feats, left_index=True, right_index=True)\n    return processed_df\n\n# ==============================================================================\n# END: Reusable Processing & Feature Extraction Functions\n# ==============================================================================\n\n\ndef run_prediction_pipeline():\n    \"\"\"\n    Main function to load the model and predict on the test set.\n    \"\"\"\n    # 1. LOAD THE TRAINED MODEL AND SCALER\n    try:\n        print(\"Loading pre-trained model and scaler...\")\n        model = joblib.load(MODEL_PATH)\n        scaler = joblib.load(SCALER_PATH)\n        print(\"Model and scaler loaded successfully.\")\n    except FileNotFoundError:\n        print(f\"[ERROR] Model or scaler not found. Please run the training script first.\")\n        return\n\n    # 2. LOAD AND PREPARE TEST EVENT MAPPING\n    try:\n        test_feedback_df = pd.read_csv(TEST_FEEDBACK_MAPPING_CSV)\n        test_feedback_df[\"_file_key\"] = test_feedback_df[FILE_COL].str.replace(r'\\.csv$', '', regex=True)\n    except FileNotFoundError:\n        print(f\"[ERROR] Test mapping file not found at: {TEST_FEEDBACK_MAPPING_CSV}\")\n        return\n\n    all_predictions = []\n    feature_cols = [f'FRN_{ch}_mean_200_300ms' for ch in ['FCz', 'Cz']] + \\\n                   ['P300_Pz_mean_300_600ms', 'Theta_4_8Hz_power_FCz_200_400ms']\n\n    # 3. LOOP THROUGH EACH TEST FILE AND MAKE PREDICTIONS\n    print(f\"\\nFound {test_feedback_df['_file_key'].nunique()} test files to process.\")\n    for file_key, group in test_feedback_df.groupby(\"_file_key\"):\n        print(f\"\\n--- Processing test file: {file_key} ---\")\n        \n        # a. Process the raw EEG file to get features using the reusable function\n        feature_df = process_single_file_to_features(file_key, group, eeg_dir=TEST_EEG_DIR)\n        \n        if feature_df is None or feature_df.empty:\n            print(f\"Could not generate features for {file_key}. Skipping.\")\n            continue\n            \n        # b. Extract feature columns in the correct order\n        X_new = feature_df[feature_cols]\n        \n        # c. Scale the new features using the LOADED scaler\n        X_new_scaled = scaler.transform(X_new)\n        \n        # d. Make predictions\n        predictions = model.predict_proba(X_new_scaled)[:, 1]\n        \n        # e. Store predictions with original info\n        result_df = feature_df.copy() # feature_df contains metadata from kept epochs\n        result_df['Prediction'] = predictions\n        all_predictions.append(result_df)\n        \n        print(f\"Successfully made {len(predictions)} predictions for this file.\")\n\n    # 4. COMBINE AND SAVE ALL PREDICTIONS\n    if all_predictions:\n        final_predictions_df = pd.concat(all_predictions, ignore_index=True)\n        \n        # Save a detailed file for your own analysis\n        final_predictions_df.to_csv(OUTPUT_PREDICTIONS_CSV, index=False)\n        print(f\"\\nDetailed predictions saved to: {OUTPUT_PREDICTIONS_CSV}\")\n\n        # Save a simple submission file (e.g., for Kaggle)\n        submission_df = final_predictions_df[['Feedback_ID', 'Prediction']]\n        submission_df.to_csv(OUTPUT_SUBMISSION_CSV, index=False)\n        print(f\"Submission file saved to: {OUTPUT_SUBMISSION_CSV}\")\n        \n        print(\"\\n--- First 5 Predictions ---\")\n        print(submission_df.head())\n    else:\n        print(\"\\nNo predictions were made. Check for warnings above.\")\n\n\nif __name__ == \"__main__\":\n    run_prediction_pipeline()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T15:36:56.537107Z","iopub.execute_input":"2025-10-03T15:36:56.537417Z","iopub.status.idle":"2025-10-03T15:46:50.219706Z","shell.execute_reply.started":"2025-10-03T15:36:56.537395Z","shell.execute_reply":"2025-10-03T15:46:50.218907Z"}},"outputs":[],"execution_count":null}]}