{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd, numpy as np\nfrom glob import glob\nimport matplotlib.pyplot as plt\nVER = 1","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:15:38.525754Z","iopub.execute_input":"2025-06-07T06:15:38.526387Z","iopub.status.idle":"2025-06-07T06:15:40.242277Z","shell.execute_reply.started":"2025-06-07T06:15:38.526356Z","shell.execute_reply":"2025-06-07T06:15:40.241371Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define where all the parquet files live\nBASE_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\n\n# Gather every parquet file recursively under BASE_PATH\nfile_paths = glob(os.path.join(BASE_PATH, '**', '*.parquet'), recursive=True)\n\n# Build a DataFrame listing each file\ndf = pd.DataFrame({'path': file_paths})\n\n# Derive the test type by taking the parent folder name after the underscore\ndf['test_type'] = df['path'].apply(\n    lambda p: os.path.basename(os.path.dirname(p)).split('_')[-1]\n)\n\n# Derive the sample ID by stripping directory and extension from the filename\ndf['id'] = df['path'].apply(\n    lambda p: os.path.splitext(os.path.basename(p))[0]\n)\n\n# Read one example parquet to verify contents\ndf_eeg = pd.read_parquet(\n    os.path.join(BASE_PATH, 'train_eegs', '1000913311.parquet')\n)\ndf_eeg.head()\n","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:15:40.244086Z","iopub.execute_input":"2025-06-07T06:15:40.244840Z","iopub.status.idle":"2025-06-07T06:16:38.050090Z","shell.execute_reply.started":"2025-06-07T06:15:40.244803Z","shell.execute_reply":"2025-06-07T06:16:38.049232Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count how many EEG channels are present:\n# Each column in df_eeg corresponds to one channel\nn_channels = len(df_eeg.columns)\n\n# Display the channel count\nn_channels\n","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:16:38.051191Z","iopub.execute_input":"2025-06-07T06:16:38.051451Z","iopub.status.idle":"2025-06-07T06:16:38.056641Z","shell.execute_reply.started":"2025-06-07T06:16:38.051429Z","shell.execute_reply":"2025-06-07T06:16:38.055795Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Specify the full path to the CSV of metadata and labels\ncsv_path = '/kaggle/input/hms-harmful-brain-activity-classification/train.csv'\n\n# Read in the training table\ndf = pd.read_csv(csv_path)\n\n# The final six columns are our classification targets\nTARGETS = df.columns[-6:]\n\n# Print a concise summary of rows/columns and list out the target names\nnum_rows, num_cols = df.shape\nprint(f\"Train.csv contains {num_rows} records across {num_cols} columns.\")\nprint(f\"Target columns: {TARGETS.tolist()}\")\n\n# Quick look at the first few rows to confirm everything loaded correctly\ndf.head()\n","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:16:38.058391Z","iopub.execute_input":"2025-06-07T06:16:38.058698Z","iopub.status.idle":"2025-06-07T06:16:38.331172Z","shell.execute_reply.started":"2025-06-07T06:16:38.058677Z","shell.execute_reply":"2025-06-07T06:16:38.330350Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 1: Extract the start of each EEG segment (per eeg_id)\n# For each eeg_id, get the first spectrogram_id and the minimum offset time\nsegment_bounds = df.groupby('eeg_id')[['spectrogram_id', 'spectrogram_label_offset_seconds']].agg({\n    'spectrogram_id': 'first',\n    'spectrogram_label_offset_seconds': 'min'\n})\nsegment_bounds.columns = ['spec_id', 'min']\n\n# Step 2: Append the end time of each EEG segment\n# Find the max offset time per eeg_id to mark the segment's end\nend_times = df.groupby('eeg_id')[['spectrogram_label_offset_seconds']].agg('max')\nsegment_bounds['max'] = end_times\n\n# Step 3: Attach patient_id per eeg_id\n# Map each EEG to its corresponding patient\npatient_info = df.groupby('eeg_id')[['patient_id']].agg('first')\nsegment_bounds['patient_id'] = patient_info\n\n# Step 4: Aggregate target label counts\n# Sum all target values per eeg_id (e.g. votes for each class)\ntarget_counts = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor label in TARGETS:\n    segment_bounds[label] = target_counts[label].values\n\n# Step 5: Normalize targets to get class probabilities\n# Convert vote counts to probability distributions\nlabel_matrix = segment_bounds[TARGETS].values\nlabel_matrix = label_matrix / label_matrix.sum(axis=1, keepdims=True)\nsegment_bounds[TARGETS] = label_matrix\n\n# Step 6: Add expert consensus label\n# Pull in the expert-assigned label per eeg_id\nexpert_labels = df.groupby('eeg_id')[['expert_consensus']].agg('first')\nsegment_bounds['target'] = expert_labels\n\n# Step 7: Finalize the training DataFrame\n# Reset index so eeg_id becomes a column instead of index\ntrain = segment_bounds.reset_index()\n\nprint('Train non-overlapp eeg_id shape:', train.shape)\ntrain.head()\n","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:16:38.332210Z","iopub.execute_input":"2025-06-07T06:16:38.332480Z","iopub.status.idle":"2025-06-07T06:16:38.420681Z","shell.execute_reply.started":"2025-06-07T06:16:38.332458Z","shell.execute_reply":"2025-06-07T06:16:38.419815Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"READ_SPEC_FILES = False # If READ_SPEC_FILES is False, the code reads the combined file instead of individual files.\nFEATURE_ENGINEER = True\n","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:16:38.421817Z","iopub.execute_input":"2025-06-07T06:16:38.422107Z","iopub.status.idle":"2025-06-07T06:16:38.426044Z","shell.execute_reply.started":"2025-06-07T06:16:38.422085Z","shell.execute_reply":"2025-06-07T06:16:38.425171Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# === Load all spectrogram parquet files from directory ===\nSPECTROGRAM_DIR = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfile_list = os.listdir(SPECTROGRAM_DIR)\nprint(f'There are {len(file_list)} spectrogram parquets')\n\nif READ_SPEC_FILES:\n    # Initialize dictionary to store loaded spectrogram data\n    spectrograms = {}\n    for idx, file_name in enumerate(file_list):\n        if idx % 100 == 0:\n            print(idx, ', ', end='')\n\n        # Read parquet, skip first column (e.g., time offset)\n        spec_df = pd.read_parquet(f'{SPECTROGRAM_DIR}{file_name}')\n        spec_id = int(file_name.split('.')[0])\n        spectrograms[spec_id] = spec_df.iloc[:, 1:].values\nelse:\n    # Load pre-processed spectrograms from .npy file\n    spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy', allow_pickle=True).item()\n","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:16:38.427187Z","iopub.execute_input":"2025-06-07T06:16:38.427449Z","iopub.status.idle":"2025-06-07T06:17:27.543437Z","shell.execute_reply.started":"2025-06-07T06:16:38.427428Z","shell.execute_reply":"2025-06-07T06:17:27.542636Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time\n\n# === Feature Extraction from Spectrograms ===\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Compute derived features from spectrogram data\n# Each spectrogram has 400 frequency channels, and we extract summary stats:\n#   - Mean and Min over a 10-minute window\n#   - Mean and Min over a 20-second window\n# This results in 1600 features per EEG sample (400 x 4)\n\nSPEC_COLS = pd.read_parquet(f'{SPECTROGRAM_DIR }1000086677.parquet').columns[1:]\n\nFEATURES = [f'{col}_mean_15m' for col in SPEC_COLS]\nFEATURES += [f'{col}_min_15m' for col in SPEC_COLS]\nFEATURES += [f'{col}_mean_50s' for col in SPEC_COLS]\nFEATURES += [f'{col}_min_50s' for col in SPEC_COLS]\n\nprint(f'We are creating {len(FEATURES)} features for {len(train)} rows... ', end='')\n\n# Initialize and populate the feature matrix\nif FEATURE_ENGINEER:\n    feature_matrix = np.zeros((len(train), len(FEATURES)))\n\n    for idx in range(len(train)):\n        if idx % 100 == 0:\n            print(idx, ', ', end='')\n\n        row_data = train.iloc[idx]\n        center_index = int((row_data['min'] + row_data['max']) // 4)\n\n        # --- 15-minute window statistics (approx. 450 time steps) ---\n        segment = spectrograms[row_data.spec_id][center_index:center_index + 450, :]\n        feature_vals = np.nanmean(segment, axis=0)\n        feature_matrix[idx, :400] = feature_vals\n        feature_vals = np.nanmin(segment, axis=0)\n        feature_matrix[idx, 400:800] = feature_vals\n\n        # --- 50-second window statistics (approx. 25 time steps) ---\n        short_segment = spectrograms[row_data.spec_id][center_index + 145:center_index + 170, :]\n        feature_vals = np.nanmean(short_segment, axis=0)\n        feature_matrix[idx, 800:1200] = feature_vals\n        feature_vals = np.nanmin(short_segment, axis=0)\n        feature_matrix[idx, 1200:1600] = feature_vals\n\n\n    # Add features to training dataframe\n    train[FEATURES] = feature_matrix\nelse:\n    # Load pre-engineered features if skipping extraction\n    train = pd.read_parquet('/kaggle/input/brain-spectrograms/train.pqt')\n\nprint()\nprint('New train shape:', train.shape)\n","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:18:26.740017Z","iopub.execute_input":"2025-06-07T06:18:26.740726Z","iopub.status.idle":"2025-06-07T06:18:45.094299Z","shell.execute_reply.started":"2025-06-07T06:18:26.740693Z","shell.execute_reply":"2025-06-07T06:18:45.093158Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy import signal\nfrom sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:18:59.781024Z","iopub.execute_input":"2025-06-07T06:18:59.781676Z","iopub.status.idle":"2025-06-07T06:19:00.641337Z","shell.execute_reply.started":"2025-06-07T06:18:59.781630Z","shell.execute_reply":"2025-06-07T06:19:00.640396Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_frequency_band_features(segment):\n    # Specify frequency ranges corresponding to common EEG bands\n    eeg_bands = {\n        'Delta': (0.5, 4),\n        'Theta': (4, 8),\n        'Alpha': (8, 12),\n        'Beta': (12, 30),\n        'Gamma': (30, 45)\n    }\n\n    band_features = []\n\n    for band in eeg_bands:\n        low, high = eeg_bands[band]\n        \n        # Design a 3rd-order bandpass filter for the current frequency band\n        bandpass_sos = signal.butter(3, [low, high], btype='bandpass', fs=200, output='sos')\n        \n        # Apply the filter to isolate the frequency content within the band\n        filtered_signal = signal.sosfilt(bandpass_sos, segment)\n        \n        # Compute statistical summaries from the filtered signal\n        band_features.extend([\n            np.nanmean(filtered_signal),   # Average amplitude\n            np.nanstd(filtered_signal),    # Signal variability\n            np.nanmax(filtered_signal),    # Peak value\n            np.nanmin(filtered_signal)     # Minimum value\n        ])\n    \n    return band_features\n","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:19:00.643037Z","iopub.execute_input":"2025-06-07T06:19:00.643663Z","iopub.status.idle":"2025-06-07T06:19:00.649819Z","shell.execute_reply.started":"2025-06-07T06:19:00.643627Z","shell.execute_reply":"2025-06-07T06:19:00.648952Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Here we comment all code in this cell is because this cell takes 30 mins and all fetures are not important(not in top 30)\n\n# import time\n# from sklearn.impute import SimpleImputer\n\n# # Initialize a PCA model\n# pca = PCA(n_components=0.95)\n# print(\"PCA model initialized.\")\n\n# # Initialize an array for original features\n# num_rows = len(train)\n# num_features = 20 * n_channels  # 20 features per channel\n# data_original = np.zeros((num_rows, num_features))\n\n# print(\"Starting feature extraction and PCA processing...\")\n# start_time = time.time()\n\n# for k in range(num_rows):\n#     if k % 1000 == 0:\n#         print(f\"Processing row {k} of {num_rows}...\")\n\n#     row = train.iloc[k]\n#     r = int((row['min'] + row['max']) // 4)\n#     eeg_segment = spectrograms[row.spec_id][r:r+300, :]\n\n#     # Apply the feature extraction function to each EEG channel\n#     all_channel_features = []\n#     for i in range(n_channels):\n#         channel_features = extract_frequency_band_features(eeg_segment[:, i])\n#         all_channel_features.extend(channel_features)\n    \n#     data_original[k, :] = all_channel_features\n\n# print(\"Data matrix constructed\")\n\n# # Impute NaN values in the data matrix\n# imputer = SimpleImputer(strategy='mean')\n# data_imputed = imputer.fit_transform(data_original)\n\n# print(f\"NaN values handled. Imputed data matrix shape: {data_imputed.shape}\")\n\n# # Apply PCA on the imputed data\n# pca.fit(data_imputed)\n# print(\"PCA fitting completed.\")\n\n# # Transform data using PCA\n# data_pca = pca.transform(data_imputed)\n\n# # Add PCA features to DataFrame\n# pca_feature_columns = [f'pca_feature_{i}' for i in range(data_pca.shape[1])]\n# train[pca_feature_columns] = data_pca\n\n# # Measure total processing time\n# total_time = time.time() - start_time\n# print(f\"Total processing time: {total_time:.2f} seconds.\")","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:19:08.609839Z","iopub.execute_input":"2025-06-07T06:19:08.610400Z","iopub.status.idle":"2025-06-07T06:19:08.614923Z","shell.execute_reply.started":"2025-06-07T06:19:08.610369Z","shell.execute_reply":"2025-06-07T06:19:08.614048Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:19:11.529024Z","iopub.execute_input":"2025-06-07T06:19:11.529731Z","iopub.status.idle":"2025-06-07T06:19:11.559191Z","shell.execute_reply.started":"2025-06-07T06:19:11.529699Z","shell.execute_reply":"2025-06-07T06:19:11.558218Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import gc\nimport xgboost as xgb\nfrom sklearn.model_selection import KFold, GroupKFold\n\nprint('XGBoost version', xgb.__version__)","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:19:14.546251Z","iopub.execute_input":"2025-06-07T06:19:14.546586Z","iopub.status.idle":"2025-06-07T06:19:15.199312Z","shell.execute_reply.started":"2025-06-07T06:19:14.546559Z","shell.execute_reply":"2025-06-07T06:19:15.198396Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import GroupKFold\nimport xgboost as xgb\nimport gc\n\n# (Assume train, FEATURES, TARGETS, TARS, VER are already defined)\n\nall_oof   = []\nall_true  = []\nall_evals = []    # <-- we’ll collect eval results here\n\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\ngkf  = GroupKFold(n_splits=5)\n\nfor i, (train_index, valid_index) in enumerate(\n        gkf.split(train, train.target, train.patient_id)\n    ):\n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n\n    model = xgb.XGBClassifier(\n        objective='multi:softprob',\n        num_class=len(TARS),\n        learning_rate=0.1,\n        tree_method='gpu_hist',  # or 'hist' if you don’t want GPU\n        eval_metric='mlogloss'\n    )\n\n    # Prepare training and validation data\n    X_train = train.loc[train_index, FEATURES]\n    y_train = train.loc[train_index, 'target'].map(TARS)\n    X_valid = train.loc[valid_index, FEATURES]\n    y_valid = train.loc[valid_index, 'target'].map(TARS)\n\n    # Fit with early stopping; capture eval results\n    model.fit(\n        X_train, y_train,\n        eval_set=[(X_valid, y_valid)],\n        verbose=False,\n        early_stopping_rounds=10\n    )\n\n    # Grab the eval history for the validation set (mlogloss per boosting round)\n    evals_result = model.evals_result()\n    # In XGBoost’s scikit‐learn API, this will be a dict like:\n    # {'validation_0': {'mlogloss': [...], 'merror': [...] (if tracked)} }\n    val_logloss = evals_result['validation_0']['mlogloss']\n    all_evals.append(val_logloss)\n\n    # Save OOF probabilities and truths\n    oof = model.predict_proba(X_valid)\n    all_oof.append(oof)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n\n    # Optionally save the model to disk\n    model.save_model(f'XGB_v{VER}_f{i}.model')\n\n    # Clean up\n    del X_train, y_train, X_valid, y_valid, oof\n    gc.collect()\n\n# Concatenate OOF arrays if you need them later\nall_oof  = np.concatenate(all_oof, axis=0)\nall_true = np.concatenate(all_true, axis=0)\n\n# -----------------------\n# Now: Plot the learning curves\n# -----------------------\nplt.figure(figsize=(8, 6))\n\nfor fold_idx, logloss_curve in enumerate(all_evals):\n    plt.plot(\n        logloss_curve,\n        label=f'Fold {fold_idx+1}',\n        linewidth=1.5\n    )\n\nplt.xlabel('Boosting Round', fontsize=12)\nplt.ylabel('Validation Log‐Loss', fontsize=12)\nplt.title('XGBoost Validation Log‐Loss vs. Boosting Round (5‐Fold)', fontsize=14)\nplt.legend(loc='upper right', fontsize=10)\nplt.grid(alpha=0.3)\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:19:15.201151Z","iopub.execute_input":"2025-06-07T06:19:15.201669Z","iopub.status.idle":"2025-06-07T06:21:47.361513Z","shell.execute_reply.started":"2025-06-07T06:19:15.201635Z","shell.execute_reply":"2025-06-07T06:21:47.360643Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TOP_K = 30\n\n# Retrieve importance scores assigned by the trained model\nimportances = model.feature_importances_\n\n# Extract feature names from the training DataFrame\nall_features = train.columns\n\n# Determine the order of features based on importance (ascending)\nranking_indices = np.argsort(importances)\n\n# Visualize the top K most important features\nplt.figure(figsize=(10, 8))\ntop_indices = ranking_indices[-TOP_K:]\nplt.barh(np.arange(TOP_K), importances[top_indices], align='center')\nplt.yticks(np.arange(TOP_K), all_features[top_indices])\nplt.title(f'Top {TOP_K} Feature Importances')\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:21:47.363010Z","iopub.execute_input":"2025-06-07T06:21:47.363273Z","iopub.status.idle":"2025-06-07T06:21:47.873340Z","shell.execute_reply.started":"2025-06-07T06:21:47.363251Z","shell.execute_reply":"2025-06-07T06:21:47.872510Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:21:47.874856Z","iopub.execute_input":"2025-06-07T06:21:47.875573Z","iopub.status.idle":"2025-06-07T06:21:47.894877Z","shell.execute_reply.started":"2025-06-07T06:21:47.875538Z","shell.execute_reply":"2025-06-07T06:21:47.893927Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"s = 853520\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\nspec = pd.read_parquet(f'{PATH2}{s}.parquet')\nspec","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:21:47.897046Z","iopub.execute_input":"2025-06-07T06:21:47.897323Z","iopub.status.idle":"2025-06-07T06:21:47.969490Z","shell.execute_reply.started":"2025-06-07T06:21:47.897300Z","shell.execute_reply":"2025-06-07T06:21:47.968641Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# FEATURE ENGINEER TEST\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\ndata = np.zeros((len(test),len(FEATURES)))\n    \nfor k in range(len(test)):\n    row = test.iloc[k]\n    s = int( row.spectrogram_id )\n    spec = pd.read_parquet(f'{PATH2}{s}.parquet')\n    \n    # 10 MINUTE WINDOW FEATURES\n    x = np.nanmean( spec.iloc[:,1:].values, axis=0)\n    data[k,:400] = x\n    x = np.nanmin( spec.iloc[:,1:].values, axis=0)\n    data[k,400:800] = x\n\n    # 20 SECOND WINDOW FEATURES\n    x = np.nanmean( spec.iloc[145:155,1:].values, axis=0)\n    data[k,800:1200] = x\n    x = np.nanmin( spec.iloc[145:155,1:].values, axis=0)\n    data[k,1200:1600] = x\n\ntest[FEATURES] = data\nprint('New test shape',test.shape)\nprint(test)","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:21:47.970623Z","iopub.execute_input":"2025-06-07T06:21:47.970945Z","iopub.status.idle":"2025-06-07T06:21:48.668417Z","shell.execute_reply.started":"2025-06-07T06:21:47.970915Z","shell.execute_reply":"2025-06-07T06:21:48.667531Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# INFER XGBOOST ON TEST\npreds = []\n\nfor i in range(5):\n    print(i, ', ', end='')\n    \n    # Load the XGBoost model\n    model = xgb.XGBClassifier()\n    model.load_model(f'XGB_v{VER}_f{i}.model')\n    \n    # Make predictions\n    pred = model.predict_proba(test[FEATURES])\n    preds.append(pred)\n\n# Average the predictions from each fold\npred = np.mean(preds, axis=0)\nprint()\nprint('Test preds shape', pred.shape)\nprint(pred)","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:21:48.669470Z","iopub.execute_input":"2025-06-07T06:21:48.669763Z","iopub.status.idle":"2025-06-07T06:21:49.479617Z","shell.execute_reply.started":"2025-06-07T06:21:48.669739Z","shell.execute_reply":"2025-06-07T06:21:49.477151Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[TARGETS] = pred\nsub.to_csv('submission.csv',index=False)\nprint('Submission shape',sub.shape)\n\nsub.head()\nprint(sub)","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:21:49.480823Z","iopub.execute_input":"2025-06-07T06:21:49.481095Z","iopub.status.idle":"2025-06-07T06:21:49.498581Z","shell.execute_reply.started":"2025-06-07T06:21:49.481072Z","shell.execute_reply":"2025-06-07T06:21:49.497647Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nsub.iloc[:,-6:].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2025-06-07T06:21:49.499753Z","iopub.execute_input":"2025-06-07T06:21:49.500015Z","iopub.status.idle":"2025-06-07T06:21:49.509894Z","shell.execute_reply.started":"2025-06-07T06:21:49.499979Z","shell.execute_reply":"2025-06-07T06:21:49.509042Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}