{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":4043,"databundleVersionId":44567}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Basic check: list all files in input directory\nimport os\n\nprint(\"Listing files and folders in /kaggle/input/ ...\")\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# Load basic CSV files to check\nimport pandas as pd\n\nlabels_df = pd.read_csv('/kaggle/input/inria-bci-challenge/TrainLabels.csv')\n\n\n# Adjust paths to your actual files inside the competition data folder\ntrain_labels = pd.read_csv('/kaggle/input/inria-bci-challenge/TrainLabels.csv')\nchannels_loc = pd.read_csv('/kaggle/input/inria-bci-challenge/ChannelsLocation.csv')\nsample_submission = pd.read_csv('/kaggle/input/inria-bci-challenge/SampleSubmission.csv')\n\nprint(\"\\nTrain labels example:\")\nprint(train_labels.head())\n\nprint(\"\\nChannels Location example:\")\nprint(channels_loc.head())\n\nprint(\"\\nSample Submission example:\")\nprint(sample_submission.head())\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-29T18:33:48.269275Z","iopub.execute_input":"2025-09-29T18:33:48.272810Z","iopub.status.idle":"2025-09-29T18:33:48.346934Z","shell.execute_reply.started":"2025-09-29T18:33:48.272710Z","shell.execute_reply":"2025-09-29T18:33:48.345583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\n\nzip_train_path = '/kaggle/input/inria-bci-challenge/train.zip'\nzip_test_path = '/kaggle/input/inria-bci-challenge/test.zip'\nextract_path = '/kaggle/working/bci_eeg_data'  # Working directory\n\n# Unzip train data\nwith zipfile.ZipFile(zip_train_path, 'r') as zip_ref:\n    zip_ref.extractall(f'{extract_path}/train')\n\n# Unzip test data\nwith zipfile.ZipFile(zip_test_path, 'r') as zip_ref:\n    zip_ref.extractall(f'{extract_path}/test')\n\nprint(\"Train and test data extracted.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T18:33:48.349062Z","iopub.execute_input":"2025-09-29T18:33:48.349370Z","iopub.status.idle":"2025-09-29T18:37:04.087954Z","shell.execute_reply.started":"2025-09-29T18:33:48.349347Z","shell.execute_reply":"2025-09-29T18:37:04.086887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfor root, dirs, files in os.walk(extract_path):\n    print(f\"Folder: {root}\")\n    for f in files[:5]:  # show first 5 files per folder for brevity\n        print(f\"  {f}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T18:37:04.089062Z","iopub.execute_input":"2025-09-29T18:37:04.089395Z","iopub.status.idle":"2025-09-29T18:37:04.096475Z","shell.execute_reply.started":"2025-09-29T18:37:04.089369Z","shell.execute_reply":"2025-09-29T18:37:04.095585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\ntrain_folder = f'{extract_path}/train'\nexample_file = os.listdir(train_folder)[0]\nexample_path = f'{train_folder}/{example_file}'\n\ndef load_eeg_csv(path):\n    df = pd.read_csv(path)\n    eeg_data = df.drop(columns=['Time', 'FeedBackEvent']).values.T  # channels x samples\n    return eeg_data\n\neeg_raw = load_eeg_csv(example_path)\nprint(f\"Loaded EEG shape (channels x samples): {eeg_raw.shape}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T18:37:04.097466Z","iopub.execute_input":"2025-09-29T18:37:04.097759Z","iopub.status.idle":"2025-09-29T18:37:07.396707Z","shell.execute_reply.started":"2025-09-29T18:37:04.097728Z","shell.execute_reply":"2025-09-29T18:37:07.395581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nchannel = 0  # Plot first channel\nplt.figure(figsize=(15, 4))\nplt.plot(eeg_raw[channel, :1500])\nplt.title('Raw EEG Signal (Channel 0) - First 1500 samples')\nplt.xlabel('Sample number')\nplt.ylabel('Amplitude')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T18:37:07.399378Z","iopub.execute_input":"2025-09-29T18:37:07.399677Z","iopub.status.idle":"2025-09-29T18:37:07.628545Z","shell.execute_reply.started":"2025-09-29T18:37:07.399649Z","shell.execute_reply":"2025-09-29T18:37:07.627692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy import signal\n\ndef preprocess_eeg(eeg_data, fs=600):\n    # Band-pass filter between 0.1 Hz and 30 Hz\n    sos = signal.butter(4, [0.1, 30], btype='band', fs=fs, output='sos')\n    filtered = signal.sosfilt(sos, eeg_data, axis=1)\n    \n    # Common Average Reference (CAR)\n    car = filtered - np.mean(filtered, axis=0, keepdims=True)\n    return car\n\neeg_processed = preprocess_eeg(eeg_raw)\nprint('Preprocessing done.')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T18:37:07.629558Z","iopub.execute_input":"2025-09-29T18:37:07.629912Z","iopub.status.idle":"2025-09-29T18:37:07.816513Z","shell.execute_reply.started":"2025-09-29T18:37:07.629884Z","shell.execute_reply":"2025-09-29T18:37:07.815666Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15, 4))\nplt.plot(eeg_processed[channel, :1500])\nplt.title('Preprocessed EEG Signal (Channel 0) - First 1500 samples')\nplt.xlabel('Sample number')\nplt.ylabel('Amplitude')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T18:37:07.817635Z","iopub.execute_input":"2025-09-29T18:37:07.817912Z","iopub.status.idle":"2025-09-29T18:37:08.027361Z","shell.execute_reply.started":"2025-09-29T18:37:07.817891Z","shell.execute_reply":"2025-09-29T18:37:08.026303Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"def extract_epochs(file_path, fs=600, epoch_duration=1.3):\n  \n    df = pd.read_csv(file_path)\n    eeg_data = df.drop(columns=['Time', 'FeedBackEvent']).values.T  # (channels, samples)\n    \n    feedback_idxs = df.index[df['FeedBackEvent'] == 1].tolist()\n    epoch_length = int(fs * epoch_duration)\n    \n    epochs = []\n    for idx in feedback_idxs:\n        if idx + epoch_length <= eeg_data.shape[1]:\n            segment = eeg_data[:, idx:idx + epoch_length]\n            segment_preprocessed = preprocess_eeg(segment, fs)\n            epochs.append(segment_preprocessed)\n    \n    return epochs\n\n# Example run on one EEG file\nexample_epochs = extract_epochs(example_path)\nprint(f\"Extracted {len(example_epochs)} feedback-aligned epochs from example file.\")\nprint(f\"Each epoch shape: {example_epochs[0].shape if example_epochs else 'N/A'}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T18:37:08.028549Z","iopub.execute_input":"2025-09-29T18:37:08.028907Z","iopub.status.idle":"2025-09-29T18:37:10.101907Z","shell.execute_reply.started":"2025-09-29T18:37:08.028880Z","shell.execute_reply":"2025-09-29T18:37:10.101067Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def reject_artifacts(epochs, thresh=200):\n    clean_epochs = []\n    for e in epochs:\n        if np.all(np.abs(e) < thresh):  # keep epoch if no channel exceeds threshold\n            clean_epochs.append(e)\n    return clean_epochs\n\nall_clean_epochs = []\nall_clean_epoch_ids = []\n\n# Iterate over all training files\nfor filename in sorted(os.listdir(train_folder)):\n    prefix = filename.replace('Data_','').replace('.csv','')\n    path = os.path.join(train_folder, filename)\n    epochs = extract_epochs(path)\n    \n    # Reject bad epochs by threshold\n    clean_epochs = reject_artifacts(epochs)\n\n    all_clean_epochs.extend(clean_epochs)\n    \n    # Keep only IDs for accepted epochs\n    for i in range(len(clean_epochs)):\n        epoch_id = f\"{prefix}_FB{i + 1:03d}\"\n        all_clean_epoch_ids.append(epoch_id)\n\nprint(f\"Extracted total clean epochs: {len(all_clean_epochs)}\")\n\n# Map labels to clean epochs\nlabels_map = labels_df.set_index('IdFeedBack')['Prediction'].to_dict()\nepoch_labels = [labels_map.get(eid, 0) for eid in all_clean_epoch_ids]\n\nprint(f\"Total labels assigned after artifact rejection: {len(epoch_labels)}\")\nprint(f\"Label distribution: {pd.Series(epoch_labels).value_counts().to_dict()}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T18:37:10.102647Z","iopub.execute_input":"2025-09-29T18:37:10.102887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\ndef extract_temporal_features(epochs, fs=600):\n    # Temporal windows means in multiple sliding windows and lags\n    window_sizes_ms = np.arange(50, 700, 50)  # 50 to 650 ms\n    lags_ms = np.arange(0, 1300, 50)          # 0 to 1250 ms\n    window_sizes = (window_sizes_ms * fs // 1000).astype(int)\n    lags = (lags_ms * fs // 1000).astype(int)\n\n    features = []\n    for epo in epochs:\n        feat_epo = []\n        for ch in range(epo.shape[0]):\n            for w in window_sizes:\n                for lag in lags:\n                    if lag + w <= epo.shape[1]:\n                        segment = epo[ch, lag:lag + w]\n                        feat_epo.append(segment.mean())\n                    else:\n                        feat_epo.append(0)\n        features.append(feat_epo)\n    return np.nan_to_num(np.array(features))\n\n\ndef extract_template_features(epochs, labels):\n    positive_epochs = [epochs[i] for i in range(len(epochs)) if labels[i] == 1]\n    if len(positive_epochs) == 0:\n        # No positive epochs, return zero array of proper shape\n        return np.zeros((len(epochs), epochs[0].shape[0]))\n    \n    template = np.mean(positive_epochs, axis=0)\n    \n    features = []\n    for epo in epochs:\n        corr_channels = []\n        for ch in range(epo.shape[0]):\n            std_epo_ch = np.std(epo[ch])\n            std_tmpl_ch = np.std(template[ch])\n            if std_epo_ch > 0 and std_tmpl_ch > 0:\n                corr = np.corrcoef(epo[ch], template[ch])[0, 1]\n            else:\n                corr = 0\n            corr_channels.append(corr)\n        features.append(corr_channels)\n    return np.nan_to_num(np.array(features))\n\n\ndef extract_statistical_features(epochs, fs=600):\n    # Statistics between 200-600 ms of each channel (mean, std, max, min, median)\n    start_idx = int(0.2 * fs)\n    end_idx = int(0.6 * fs)\n\n    features = []\n    for epo in epochs:\n        feat_epo = []\n        for ch in range(epo.shape[0]):\n            segment = epo[ch, start_idx:end_idx] if end_idx <= epo.shape[1] else epo[ch]\n            feat_epo.extend([\n                segment.mean(),\n                segment.std(),\n                segment.max(),\n                segment.min(),\n                np.median(segment)\n            ])\n        features.append(feat_epo)\n    return np.array(features)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import StratifiedKFold, cross_val_score\n\n# Extract features\nX_temp = extract_temporal_features(all_clean_epochs)\nX_tmpl = extract_template_features(all_clean_epochs, epoch_labels)\nX_stat = extract_statistical_features(all_clean_epochs)\n\nprint(\"Feature shapes:\", X_temp.shape, X_tmpl.shape, X_stat.shape)\n\n# Normalize\nscaler_temp = StandardScaler()\nscaler_tmpl = StandardScaler()\nscaler_stat = StandardScaler()\n\nX_temp_norm = scaler_temp.fit_transform(X_temp)\nX_tmpl_norm = scaler_tmpl.fit_transform(X_tmpl)\nX_stat_norm = scaler_stat.fit_transform(X_stat)\n\n# Prepare feature sets for two SVMs\nX1 = np.hstack([X_temp_norm, X_stat_norm])\nX2 = np.hstack([X_tmpl_norm, X_stat_norm])\n\ny_array = np.array(epoch_labels)\n\n# Train and validate with 5-fold Stratified CV\ncv = StratifiedKFold(n_splits=5, shuffle=True, random_state=42)\nsvm1 = SVC(kernel='linear', probability=True)\nsvm2 = SVC(kernel='linear', probability=True)\n\nauc1 = cross_val_score(svm1, X1, y_array, cv=cv, scoring='roc_auc')\nauc2 = cross_val_score(svm2, X2, y_array, cv=cv, scoring='roc_auc')\n\nprint(f\"Linear SVM (Temporal+Stats) AUC: {auc1.mean():.3f} ± {auc1.std():.3f}\")\nprint(f\"Linear SVM (Template+Stats) AUC: {auc2.mean():.3f} ± {auc2.std():.3f}\")\n\n# Fit models on full data for test prediction later\nsvm1.fit(X1, y_array)\nsvm2.fit(X2, y_array)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_folder = f'{extract_path}/test'\n\nall_test_epochs = []\nall_test_epoch_ids = []\n\n# Extract epochs from each test EEG file\nfor filename in sorted(os.listdir(test_folder)):\n    prefix = filename.replace('Data_','').replace('.csv','')\n    path = os.path.join(test_folder, filename)\n    epochs = extract_epochs(path)\n    \n    all_test_epochs.extend(epochs)\n    for i in range(len(epochs)):\n        epoch_id = f\"{prefix}_FB{i + 1:03d}\"\n        all_test_epoch_ids.append(epoch_id)\n\nprint(f\"Extracted {len(all_test_epochs)} epochs from test data\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_test_temp = extract_temporal_features(all_test_epochs)\nX_test_tmpl = extract_template_features(all_test_epochs, [0]*len(all_test_epochs))  # unlabeled, default zeros\nX_test_stat = extract_statistical_features(all_test_epochs)\n\n# Normalize using training scalers\nX_test_temp_norm = scaler_temp.transform(X_test_temp)\nX_test_tmpl_norm = scaler_tmpl.transform(X_test_tmpl)\nX_test_stat_norm = scaler_stat.transform(X_test_stat)\n\n# Combine for prediction\nX_test_1 = np.hstack([X_test_temp_norm, X_test_stat_norm])\nX_test_2 = np.hstack([X_test_tmpl_norm, X_test_stat_norm])\n\nprobs1 = svm1.predict_proba(X_test_1)[:, 1]  # probability of class 1\nprobs2 = svm2.predict_proba(X_test_2)[:, 1]\n\n# Average ensemble probabilities\nfinal_probs = (probs1 + probs2) / 2\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}