{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <div style=\"padding: 30px;color:white;margin:10;font-size:60%;text-align:left;display:fill;border-radius:10px;background-color:#FFFFFF;overflow:hidden;background-color:#FFCE30\"><b><span style='color:#FFFFFF'>2 |</span></b> <b>REFERENCE & ACKNOWLEDGEMENT</b></div>\n\nThis notebook wouldn't be possible without the valuable insights and contributions from the Kaggle community. I've leveraged several resources to compile the most effective learning path for us:\n\n* https://www.kaggle.com/code/cdeotte/catboost-starter-lb-0-8\n* https://www.kaggle.com/code/mvvppp/hms-eda-and-domain-journey\n* https://www.kaggle.com/code/ksooklall/hms-banana-montage\n* https://www.kaggle.com/code/mpwolke/seizures-classification-parquet\n\n\nFeel free to explore these resources alongside this notebook to deepen your understanding.","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd, numpy as np\nfrom glob import glob\nimport matplotlib.pyplot as plt\nVER = 2","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:41:47.217609Z","iopub.execute_input":"2025-11-04T18:41:47.218067Z","iopub.status.idle":"2025-11-04T18:41:47.644686Z","shell.execute_reply.started":"2025-11-04T18:41:47.218033Z","shell.execute_reply":"2025-11-04T18:41:47.643231Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# check the reading of one parquet for understanding\n\nBASE_PATH = '/kaggle/input/hms-harmful-brain-activity-classification/'\n\ndf = pd.DataFrame({'path': glob(BASE_PATH + '**/*.parquet')})\ndf['test_type'] = df['path'].str.split('/').str.get(-2).str.split('_').str.get(-1)\ndf['id'] = df['path'].str.split('/').str.get(-1).str.split('.').str.get(0)\n\ndf_eeg = pd.read_parquet(BASE_PATH + 'train_eegs/1000913311.parquet')\ndf_eeg.head()","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:41:47.647411Z","iopub.execute_input":"2025-11-04T18:41:47.647977Z","iopub.status.idle":"2025-11-04T18:41:48.601355Z","shell.execute_reply.started":"2025-11-04T18:41:47.647947Z","shell.execute_reply":"2025-11-04T18:41:48.600482Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Determine the number of channels\n# Assuming each row is a time point and each column is a channel\nn_channels = df_eeg.shape[1]\nn_channels","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:41:48.602319Z","iopub.execute_input":"2025-11-04T18:41:48.602556Z","iopub.status.idle":"2025-11-04T18:41:48.610270Z","shell.execute_reply.started":"2025-11-04T18:41:48.602534Z","shell.execute_reply":"2025-11-04T18:41:48.609172Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:41:48.612458Z","iopub.execute_input":"2025-11-04T18:41:48.612768Z","iopub.status.idle":"2025-11-04T18:41:48.856974Z","shell.execute_reply.started":"2025-11-04T18:41:48.612742Z","shell.execute_reply":"2025-11-04T18:41:48.856198Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Creating a Unique EEG Segment per eeg_id:\n# The code groups (groupby) the EEG data (df) by eeg_id. Each eeg_id represents a different EEG recording.\n# It then picks the first spectrogram_id and the earliest (min) spectrogram_label_offset_seconds for each eeg_id. This helps in identifying the starting point of each EEG segment.\n# The resulting DataFrame train has columns spec_id (first spectrogram_id) and min (earliest spectrogram_label_offset_seconds).\ntrain = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\n\n# Finding the Latest Point in Each EEG Segment:\n# The code again groups the data by eeg_id and finds the latest (max) spectrogram_label_offset_seconds for each segment.\n# This max value is added to the train DataFrame, representing the end point of each EEG segment.\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first') # The code adds the patient_id for each eeg_id to the train DataFrame. This links each EEG segment to a specific patient.\ntrain['patient_id'] = tmp\n\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum') # The code sums up the target variable counts (like votes for seizure, LPD, etc.) for each eeg_id.\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values # It then normalizes these counts so that they sum up to 1. This step converts the counts into probabilities, which is a common practice in classification tasks.\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first') # For each eeg_id, the code includes the expert_consensus on the EEG segment's classification.\ntrain['target'] = tmp\n\ntrain = train.reset_index() # This makes eeg_id a regular column, making the DataFrame easier to work with.\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:41:48.858035Z","iopub.execute_input":"2025-11-04T18:41:48.858288Z","iopub.status.idle":"2025-11-04T18:41:48.951647Z","shell.execute_reply.started":"2025-11-04T18:41:48.858268Z","shell.execute_reply":"2025-11-04T18:41:48.950869Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"READ_SPEC_FILES = False # If READ_SPEC_FILES is False, the code reads the combined file instead of individual files.\nFEATURE_ENGINEER = True","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:41:48.952545Z","iopub.execute_input":"2025-11-04T18:41:48.952759Z","iopub.status.idle":"2025-11-04T18:41:48.957532Z","shell.execute_reply.started":"2025-11-04T18:41:48.952740Z","shell.execute_reply":"2025-11-04T18:41:48.955950Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n# READ ALL SPECTROGRAMS\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\nfiles = os.listdir(PATH)\nprint(f'There are {len(files)} spectrogram parquets')\n\nif READ_SPEC_FILES:    \n    spectrograms = {}\n    for i,f in enumerate(files):\n        if i%100==0: print(i,', ',end='')\n        tmp = pd.read_parquet(f'{PATH}{f}')\n        name = int(f.split('.')[0])\n        spectrograms[name] = tmp.iloc[:,1:].values\nelse:\n    spectrograms = np.load('/kaggle/input/brain-spectrograms/specs.npy',allow_pickle=True).item()","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:41:48.962337Z","iopub.execute_input":"2025-11-04T18:41:48.962795Z","iopub.status.idle":"2025-11-04T18:42:48.433551Z","shell.execute_reply.started":"2025-11-04T18:41:48.962756Z","shell.execute_reply":"2025-11-04T18:42:48.431438Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%time\n# ENGINEER FEATURES\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# The code generates features from the spectrogram data for use in a model \n# The features are derived by calculating the mean and minimum values over time for each of the 400 spectrogram frequencies.\n# Two types of windows are used for these calculations:\n# A 10-minute window (_mean_10m, _min_10m).\n# A 20-second window (_mean_20s, _min_20s).\n# This process results in 1600 features (400 features × 4 calculations) for each EEG ID.\n\nSPEC_COLS = pd.read_parquet(f'{PATH}1000086677.parquet').columns[1:]\nFEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\nFEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\nFEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\nprint(f'We are creating {len(FEATURES)} features for {len(train)} rows... ',end='')\n\n\n# A data matrix data is initialized to store the new features for each eeg_id in the train DataFrame.\n# For each row in train, the code calculates the mean and minimum values within the specified 10-minute and 20-second windows.\n# These calculated values are then stored in the data matrix.\n# Finally, the matrix is added to the train DataFrame as new columns.\n\nif FEATURE_ENGINEER:\n    data = np.zeros((len(train),len(FEATURES)))\n    for k in range(len(train)):\n        if k%100==0: print(k,', ',end='')\n        row = train.iloc[k]\n        r = int( (row['min'] + row['max'])//4 ) \n        \n        # 10 MINUTE WINDOW FEATURES (MEANS and MINS)\n        x = np.nanmean(spectrograms[row.spec_id][r:r+300,:],axis=0)\n        data[k,:400] = x\n        x = np.nanmin(spectrograms[row.spec_id][r:r+300,:],axis=0)\n        data[k,400:800] = x\n        \n        # 20 SECOND WINDOW FEATURES (MEANS and MINS)\n        x = np.nanmean(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n        data[k,800:1200] = x\n        x = np.nanmin(spectrograms[row.spec_id][r+145:r+155,:],axis=0)\n        data[k,1200:1600] = x\n\n    train[FEATURES] = data\nelse:\n    train = pd.read_parquet('/kaggle/input/brain-spectrograms/train.pqt')\nprint()\nprint('New train shape:',train.shape)","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:42:48.436732Z","iopub.execute_input":"2025-11-04T18:42:48.437405Z","iopub.status.idle":"2025-11-04T18:43:05.315129Z","shell.execute_reply.started":"2025-11-04T18:42:48.437371Z","shell.execute_reply":"2025-11-04T18:43:05.313922Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CÉLULA 8.1 — Wavelets (PyWavelets) e utilitários\nimport pywt\nfrom scipy.stats import entropy\n\nFS = 200  # frequência de amostragem típica do HMS EEG\nEEG_BASE = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\n\n# Mapeamento esperado para db4 nível 5 @ FS=200 Hz:\n# D1: 50–100 (≈ruído alta freq), D2: 25–50 (Gama alta), D3: 12.5–25 (Beta),\n# D4: 6.25–12.5 (Alpha), D5: 3.125–6.25 (Theta), A5: 0–3.125 (Delta)\n\ndef dwt_band_energies(x, wavelet='db4', level=5):\n    \"\"\"Retorna energias por banda (Delta..Gamma) + entropia para um canal 1D.\"\"\"\n    coeffs = pywt.wavedec(x, wavelet=wavelet, level=level, mode='symmetric')\n    # coeffs = [A5, D5, D4, D3, D2, D1]\n    A5, D5, D4, D3, D2, D1 = coeffs\n\n    band_coeffs = {\n        'Delta': A5,        # ~0–3.1 Hz\n        'Theta': D5,        # ~3.1–6.25 Hz\n        'Alpha': D4,        # ~6.25–12.5 Hz\n        'Beta':  D3,        # ~12.5–25 Hz\n        'Gamma': D2,        # ~25–50 Hz (observação: D1 ~50–100 Hz fica de fora por robustez)\n    }\n    # energia por banda\n    energies = {b: float(np.sum(np.square(c))) for b, c in band_coeffs.items()}\n    total_energy = sum(energies.values()) + 1e-12\n    # energias relativas (robustas a escala)\n    rel = {f'{b}_relE': energies[b] / total_energy for b in energies}\n\n    # estatísticas dos coeficientes por banda\n    stats = {}\n    for b, c in band_coeffs.items():\n        c = np.asarray(c)\n        stats.update({\n            f'{b}_mean': float(np.mean(c)),\n            f'{b}_std':  float(np.std(c)),\n            f'{b}_mad':  float(np.mean(np.abs(c - np.mean(c)))),\n        })\n\n    # entropia espectral (a partir das energias relativas)\n    p = np.array(list(rel.values()), dtype=float)\n    stats['wavelet_entropy'] = float(entropy(p, base=np.e))\n\n    # pacote final por canal\n    feats = {}\n    feats.update(rel)\n    feats.update(stats)\n    return feats\n\ndef extract_wavelet_features_from_eeg(eeg_df, seconds=60, fs=FS):\n    \"\"\"\n    Recebe o DataFrame do EEG (colunas=canais), corta uma janela de 'seconds' centrada,\n    e extrai features DWT por canal. Retorna dict {feature_name: value}.\n    \"\"\"\n    x = eeg_df.values  # shape: [N_amostras, N_canais]\n    n = len(x)\n    win = int(seconds * fs)\n    if n < win:\n        # padding simples se sinal for curto\n        pad = win - n\n        x = np.pad(x, ((pad//2, pad - pad//2), (0, 0)), mode='edge')\n        n = len(x)\n    start = (n - win) // 2\n    seg = x[start:start+win, :]  # [win, n_channels]\n\n    feats = {}\n    n_channels = seg.shape[1]\n    for ch in range(n_channels):\n        ch_sig = seg[:, ch]\n        ch_feats = dwt_band_energies(ch_sig)\n        for k, v in ch_feats.items():\n            feats[f'ch{ch}_{k}'] = v\n\n    # versões agregadas (média por banda entre canais) para reduzir dimensionalidade\n    bands = ['Delta','Theta','Alpha','Beta','Gamma']\n    agg = {}\n    for key in ['relE','mean','std','mad']:\n        for b in bands:\n            cols = [f'ch{ch}_{b}_{key}' for ch in range(n_channels)]\n            vals = [feats[c] for c in cols if c in feats]\n            if vals:\n                agg[f'agg_{b}_{key}_mean'] = float(np.mean(vals))\n                agg[f'agg_{b}_{key}_std']  = float(np.std(vals))\n\n    # entropia média/STD entre canais\n    ent_cols = [f'ch{ch}_wavelet_entropy' for ch in range(n_channels)]\n    ent_vals = [feats[c] for c in ent_cols if c in feats]\n    if ent_vals:\n        agg['agg_wavelet_entropy_mean'] = float(np.mean(ent_vals))\n        agg['agg_wavelet_entropy_std']  = float(np.std(ent_vals))\n\n    # retorna tanto por-canal (mais rico) quanto agregados (mais compacto)\n    feats.update(agg)\n    return feats\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-04T19:24:50.497268Z","iopub.execute_input":"2025-11-04T19:24:50.497715Z","iopub.status.idle":"2025-11-04T19:24:50.515111Z","shell.execute_reply.started":"2025-11-04T19:24:50.497689Z","shell.execute_reply":"2025-11-04T19:24:50.513923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from scipy import signal\nfrom sklearn.decomposition import PCA","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:43:05.316723Z","iopub.execute_input":"2025-11-04T18:43:05.317089Z","iopub.status.idle":"2025-11-04T18:43:07.076012Z","shell.execute_reply.started":"2025-11-04T18:43:05.317062Z","shell.execute_reply":"2025-11-04T18:43:07.075115Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_frequency_band_features(segment):\n    # Define EEG frequency bands\n    eeg_bands = {'Delta': (0.5, 4), 'Theta': (4, 8), 'Alpha': (8, 12), 'Beta': (12, 30), 'Gamma': (30, 45)}\n    \n    band_features = []\n    for band in eeg_bands:\n        low, high = eeg_bands[band]\n        # Filter signal for the specific band\n        band_pass_filter = signal.butter(3, [low, high], btype='bandpass', fs=200, output='sos')\n        filtered = signal.sosfilt(band_pass_filter, segment)\n        # Extract features like mean, standard deviation, etc.\n        band_features.extend([np.nanmean(filtered), np.nanstd(filtered), np.nanmax(filtered), np.nanmin(filtered)])\n    \n    return band_features","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:43:07.077184Z","iopub.execute_input":"2025-11-04T18:43:07.077504Z","iopub.status.idle":"2025-11-04T18:43:07.084946Z","shell.execute_reply.started":"2025-11-04T18:43:07.077481Z","shell.execute_reply":"2025-11-04T18:43:07.083778Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import time\n# from sklearn.impute import SimpleImputer\n\n# # Initialize a PCA model\n# pca = PCA(n_components=0.95)\n# print(\"PCA model initialized.\")\n\n# # Initialize an array for original features\n# num_rows = len(train)\n# num_features = 20 * n_channels  # 20 features per channel\n# data_original = np.zeros((num_rows, num_features))\n\n# print(\"Starting feature extraction and PCA processing...\")\n# start_time = time.time()\n\n# for k in range(num_rows):\n#     if k % 1000 == 0:\n#         print(f\"Processing row {k} of {num_rows}...\")\n\n#     row = train.iloc[k]\n#     r = int((row['min'] + row['max']) // 4)\n#     eeg_segment = spectrograms[row.spec_id][r:r+300, :]\n\n#     # Apply the feature extraction function to each EEG channel\n#     all_channel_features = []\n#     for i in range(n_channels):\n#         channel_features = extract_frequency_band_features(eeg_segment[:, i])\n#         all_channel_features.extend(channel_features)\n    \n#     data_original[k, :] = all_channel_features\n\n# print(\"Data matrix constructed\")\n\n# # Impute NaN values in the data matrix\n# imputer = SimpleImputer(strategy='mean')\n# data_imputed = imputer.fit_transform(data_original)\n\n# print(f\"NaN values handled. Imputed data matrix shape: {data_imputed.shape}\")\n\n# # Apply PCA on the imputed data\n# pca.fit(data_imputed)\n# print(\"PCA fitting completed.\")\n\n# # Transform data using PCA\n# data_pca = pca.transform(data_imputed)\n\n# # Add PCA features to DataFrame\n# pca_feature_columns = [f'pca_feature_{i}' for i in range(data_pca.shape[1])]\n# train[pca_feature_columns] = data_pca\n\n# # Measure total processing time\n# total_time = time.time() - start_time\n# print(f\"Total processing time: {total_time:.2f} seconds.\")","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:43:07.086524Z","iopub.execute_input":"2025-11-04T18:43:07.086920Z","iopub.status.idle":"2025-11-04T18:43:07.100665Z","shell.execute_reply.started":"2025-11-04T18:43:07.086890Z","shell.execute_reply":"2025-11-04T18:43:07.099614Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CÉLULA 11.1 — Extração de features Wavelet (DWT) para o TRAIN\n# Estratégia: para cada eeg_id em 'train', abre o .parquet bruto e extrai features DWT (60s janela).\n# Armazenamos como colunas em 'train' e montamos FEATURES_WAVELET.\n\nfrom tqdm import tqdm\n\nFEATURES_WAVELET = None\nwavelet_rows = []\n\nprint('Extraindo wavelet features (DWT) para TRAIN...')\nfor k in tqdm(range(len(train))):\n    eeg_id = int(train.iloc[k]['eeg_id'])\n    eeg_path = os.path.join(EEG_BASE, f'{eeg_id}.parquet')\n    if not os.path.exists(eeg_path):\n        # fallback: sem arquivo (raro), cria zeros\n        wavelet_rows.append({})\n        continue\n    eeg_df = pd.read_parquet(eeg_path)\n    # remove colunas não-numéricas, se houver\n    eeg_df = eeg_df.select_dtypes(include=[np.number])\n    feats = extract_wavelet_features_from_eeg(eeg_df, seconds=60, fs=FS)\n    wavelet_rows.append(feats)\n\n# alinhar para DataFrame\nwavelet_df = pd.DataFrame(wavelet_rows).fillna(0.0)\nif FEATURES_WAVELET is None:\n    FEATURES_WAVELET = list(wavelet_df.columns)\n\nprint(f'Wavelet features geradas: {len(FEATURES_WAVELET)}')\n# anexar ao train\ntrain = pd.concat([train.reset_index(drop=True), wavelet_df.reset_index(drop=True)], axis=1)\nprint('Train+Wavelet shape:', train.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-04T19:24:56.309992Z","iopub.execute_input":"2025-11-04T19:24:56.310464Z","iopub.status.idle":"2025-11-04T19:40:27.912767Z","shell.execute_reply.started":"2025-11-04T19:24:56.310438Z","shell.execute_reply":"2025-11-04T19:40:27.911627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2025-11-04T19:41:23.738976Z","iopub.execute_input":"2025-11-04T19:41:23.739426Z","iopub.status.idle":"2025-11-04T19:41:23.768564Z","shell.execute_reply.started":"2025-11-04T19:41:23.739396Z","shell.execute_reply":"2025-11-04T19:41:23.767063Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.preprocessing import StandardScaler\n\n# # Columns to be excluded from scaling\n# excluded_columns = ['eeg_id', 'spec_id', 'min', 'max', 'patient_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote','target']\n\n# # Save the columns to be excluded\n# excluded_data = train[excluded_columns]\n\n# # DataFrame with only the columns to be scaled\n# features = train.drop(columns=excluded_columns)\n\n# # Initialize the StandardScaler\n# scaler = StandardScaler()\n\n# # Fit the scaler to the features and transform them\n# features_scaled = scaler.fit_transform(features)\n\n# # Create a DataFrame from the scaled features\n# features_scaled_df = pd.DataFrame(features_scaled, columns=features.columns)\n\n# # Concatenate the scaled features with the excluded columns\n# train_scaled_df = pd.concat([excluded_data.reset_index(drop=True),features_scaled_df,], axis=1)\n# train_scaled_df \n","metadata":{"execution":{"iopub.status.busy":"2025-11-04T19:41:27.532148Z","iopub.execute_input":"2025-11-04T19:41:27.532927Z","iopub.status.idle":"2025-11-04T19:41:27.538682Z","shell.execute_reply.started":"2025-11-04T19:41:27.532901Z","shell.execute_reply":"2025-11-04T19:41:27.537280Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import xgboost as xgb\nimport gc\nfrom sklearn.model_selection import KFold, GroupKFold\n\nprint('XGBoost version', xgb.__version__)","metadata":{"execution":{"iopub.status.busy":"2025-11-04T19:41:41.541441Z","iopub.execute_input":"2025-11-04T19:41:41.541991Z","iopub.status.idle":"2025-11-04T19:41:41.549310Z","shell.execute_reply.started":"2025-11-04T19:41:41.541954Z","shell.execute_reply":"2025-11-04T19:41:41.547595Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_oof = []\nall_true = []\nTARS = {'Seizure':0, 'LPD':1, 'GPD':2, 'LRDA':3, 'GRDA':4, 'Other':5}\n\ngkf = GroupKFold(n_splits=5)\nfor i, (train_index, valid_index) in enumerate(gkf.split(train , train .target, train .patient_id)):   \n    \n    print('#'*25)\n    print(f'### Fold {i+1}')\n    print(f'### train size {len(train_index)}, valid size {len(valid_index)}')\n    print('#'*25)\n    \n    model = xgb.XGBClassifier(\n        objective='multi:softprob', \n        num_class=len(TARS),\n        learning_rate = 0.1, \n                      \n#         tree_method='gpu_hist',  #skip GPU acceleration\n    )\n    \n    # Prepare training and validation data\n    X_train = train.loc[train_index, FEATURES]\n    y_train = train.loc[train_index, 'target'].map(TARS)\n    X_valid = train.loc[valid_index, FEATURES]\n    y_valid = train.loc[valid_index, 'target'].map(TARS)\n    \n    model.fit(X_train, y_train, \n              eval_set=[(X_valid, y_valid)], \n              verbose=True, \n              early_stopping_rounds=10)\n    model.save_model(f'XGB_v{VER}_f{i}.model')\n    \n    oof = model.predict_proba(X_valid)\n    all_oof.append(oof)\n    all_true.append(train.loc[valid_index, TARGETS].values)\n    \n    del X_train, y_train, X_valid, y_valid, oof\n    gc.collect()\n    \nall_oof = np.concatenate(all_oof)\nall_true = np.concatenate(all_true)","metadata":{"execution":{"iopub.status.busy":"2025-11-04T19:41:46.810790Z","iopub.execute_input":"2025-11-04T19:41:46.811135Z","iopub.status.idle":"2025-11-04T19:59:10.946115Z","shell.execute_reply.started":"2025-11-04T19:41:46.811114Z","shell.execute_reply":"2025-11-04T19:59:10.943987Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CÉLULA 14.3 — Teste local (holdout por patient_id, sem mudanças no treino)\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.model_selection import GroupShuffleSplit\nimport xgboost as xgb\n\ndef soft_log_loss(y_true, y_pred, eps=1e-12):\n    \"\"\"\n    Log loss (cross-entropy) para alvos probabilísticos (soft targets).\n    y_true: (N, C) com linhas somando ≈1\n    y_pred: (N, C) probabilidades preditas\n    \"\"\"\n    y_pred = np.clip(y_pred, eps, 1.0 - eps)\n    y_pred = y_pred / y_pred.sum(axis=1, keepdims=True)\n    y_true = y_true / (y_true.sum(axis=1, keepdims=True) + eps)\n    return float(-(y_true * np.log(y_pred)).sum(axis=1).mean())\n\n# 1) Seleciona 20% dos pacientes para teste local (holdout estrito por patient_id)\ngss = GroupShuffleSplit(n_splits=1, test_size=0.2, random_state=42)\nidx_train_local, idx_test_local = next(gss.split(train, groups=train['patient_id']))\n\ntrain_local = train.iloc[idx_train_local].reset_index(drop=True)\ntest_local  = train.iloc[idx_test_local].reset_index(drop=True)\n\nprint(f'Holdout local — pacientes únicos: '\n      f'train={train_local[\"patient_id\"].nunique()} | test={test_local[\"patient_id\"].nunique()}')\nprint(f'Linhas: train={len(train_local)} | test={len(test_local)}')\n\n# 2) Define colunas de features (usa wavelets se existirem)\nfeature_cols = FEATURES_ALL if 'FEATURES_ALL' in globals() else FEATURES\nprint(f'Usando {len(feature_cols)} features.')\n\nX_test_local = test_local[feature_cols]\ny_true_soft  = test_local[TARGETS].values  # alvos probabilísticos\ny_true_hard  = np.argmax(y_true_soft, axis=1)\n\n# 3) Carrega os 5 modelos salvos (um por fold) e faz ensemble (média)\npreds = []\nfor i in range(5):\n    model = xgb.XGBClassifier()\n    model.load_model(f'XGB_v{VER}_f{i}.model')\n    preds.append(model.predict_proba(X_test_local))\ny_pred = np.mean(preds, axis=0)\n\n# 4) Métricas no teste local\nll_soft = soft_log_loss(y_true_soft, y_pred)\nacc     = accuracy_score(y_true_hard, np.argmax(y_pred, axis=1))\n\nprint(f'\\n=== Resultados no TESTE LOCAL (holdout 20% por paciente) ===')\nprint(f'LogLoss (soft targets): {ll_soft:.6f}')\nprint(f'Accuracy (top-1):       {acc:.4f}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-04T20:47:27.231780Z","iopub.execute_input":"2025-11-04T20:47:27.234588Z","iopub.status.idle":"2025-11-04T20:47:29.648598Z","shell.execute_reply.started":"2025-11-04T20:47:27.234472Z","shell.execute_reply":"2025-11-04T20:47:29.647407Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CÉLULA 14.3B — Holdout local sem vazamento (modelo treinado só em train_local)\n\nimport numpy as np\nfrom sklearn.model_selection import GroupShuffleSplit\nfrom sklearn.metrics import accuracy_score\nimport xgboost as xgb\n\ndef soft_log_loss(y_true, y_pred, eps=1e-12):\n    y_pred = np.clip(y_pred, eps, 1.0 - eps)\n    y_pred = y_pred / y_pred.sum(axis=1, keepdims=True)\n    y_true = y_true / (y_true.sum(axis=1, keepdims=True) + eps)\n    return float(-(y_true * np.log(y_pred)).sum(axis=1).mean())\n\n# Usa os mesmos objetos criados na 14.3:\n# train_local, test_local, feature_cols, TARGETS\n\n# 1) split interno de validação DENTRO de train_local (por paciente) para early stopping\ngss_inner = GroupShuffleSplit(n_splits=1, test_size=0.1, random_state=7)\nidx_tr, idx_va = next(gss_inner.split(train_local, groups=train_local['patient_id']))\n\ntr = train_local.iloc[idx_tr].reset_index(drop=True)\nva = train_local.iloc[idx_va].reset_index(drop=True)\n\nX_tr, y_tr = tr[feature_cols], tr['target'].map({'Seizure':0,'LPD':1,'GPD':2,'LRDA':3,'GRDA':4,'Other':5})\nX_va, y_va = va[feature_cols], va['target'].map({'Seizure':0,'LPD':1,'GPD':2,'LRDA':3,'GRDA':4,'Other':5})\n\n# 2) treina um ÚNICO modelo só com train_local\nmodel_holdout = xgb.XGBClassifier(\n    objective='multi:softprob',\n    num_class=6,\n    learning_rate=0.1,\n    # tree_method='gpu_hist',  # habilite se tiver GPU\n    eval_metric='mlogloss',\n)\nmodel_holdout.fit(\n    X_tr, y_tr,\n    eval_set=[(X_va, y_va)],\n    verbose=True,\n    early_stopping_rounds=20\n)\n\n# 3) avalia no test_local (sem vazamento)\nX_te = test_local[feature_cols]\ny_true_soft = test_local[TARGETS].values\ny_true_hard = np.argmax(y_true_soft, axis=1)\n\ny_pred = model_holdout.predict_proba(X_te)\nll_soft = soft_log_loss(y_true_soft, y_pred)\nacc     = accuracy_score(y_true_hard, np.argmax(y_pred, axis=1))\n\nprint('\\n=== TESTE LOCAL (sem vazamento) ===')\nprint(f'LogLoss (soft targets): {ll_soft:.6f}')\nprint(f'Accuracy (top-1):       {acc:.4f}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-04T20:44:12.728885Z","iopub.execute_input":"2025-11-04T20:44:12.729289Z","iopub.status.idle":"2025-11-04T20:47:20.230405Z","shell.execute_reply.started":"2025-11-04T20:44:12.729259Z","shell.execute_reply":"2025-11-04T20:47:20.229659Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CÉLULA 14.1 (corrigida) — Métricas OOF com alvos probabilísticos\n\nimport numpy as np\nfrom sklearn.metrics import accuracy_score, log_loss\n\ndef soft_log_loss(y_true, y_pred, eps=1e-12):\n    \"\"\"\n    Log loss (cross-entropy) para alvos probabilísticos (soft targets).\n    y_true: (N, C) com linhas somando ≈1\n    y_pred: (N, C) com probabilidades (não precisa estar perfeitamente normalizado)\n    \"\"\"\n    y_pred = np.clip(y_pred, eps, 1.0 - eps)\n    y_pred = y_pred / y_pred.sum(axis=1, keepdims=True)\n    y_true = y_true / (y_true.sum(axis=1, keepdims=True) + eps)\n    return float(-(y_true * np.log(y_pred)).sum(axis=1).mean())\n\n# 1) LogLoss principal com soft targets\noof_logloss = soft_log_loss(all_true, all_oof)\nprint(f'OOF LogLoss (multiclasse, soft targets): {oof_logloss:.6f}')\n\n# 2) Diagnóstico: Accuracy top-1 (usando rótulo duro por argmax)\ntrue_hard = np.argmax(all_true, axis=1)\npred_hard = np.argmax(all_oof, axis=1)\noof_acc = accuracy_score(true_hard, pred_hard)\nprint(f'OOF Accuracy (top-1, diagnóstico): {oof_acc:.4f}')\n\n# 3) (Opcional) LogLoss \"padrão\" do sklearn com rótulo duro\n#    Útil para comparar, mas não substitui o soft log loss oficial do desafio.\noof_logloss_hard = log_loss(true_hard, all_oof)\nprint(f'OOF LogLoss (sklearn, usando rótulo duro): {oof_logloss_hard:.6f}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-04T20:20:56.702889Z","iopub.execute_input":"2025-11-04T20:20:56.703509Z","iopub.status.idle":"2025-11-04T20:20:56.737485Z","shell.execute_reply.started":"2025-11-04T20:20:56.703462Z","shell.execute_reply":"2025-11-04T20:20:56.734132Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CÉLULA 14.2 — Accuracy no treino (validação cruzada)\nfrom sklearn.metrics import accuracy_score\nimport numpy as np\n\n# all_oof = predições das folds de validação\n# all_true = alvos probabilísticos originais (votos normalizados)\n\n# converte para rótulo “duro” (classe mais votada)\ntrue_labels = np.argmax(all_true, axis=1)\npred_labels = np.argmax(all_oof, axis=1)\n\nacc_train = accuracy_score(true_labels, pred_labels)\nprint(f'Accuracy média (OOF - treino cruzado): {acc_train:.4f}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-04T20:15:35.905013Z","iopub.execute_input":"2025-11-04T20:15:35.905532Z","iopub.status.idle":"2025-11-04T20:15:35.916207Z","shell.execute_reply.started":"2025-11-04T20:15:35.905506Z","shell.execute_reply":"2025-11-04T20:15:35.914458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import optuna\n# from sklearn.metrics import log_loss\n\n\n# def objective(trial):\n#     # Hyperparameters to be tuned by Optuna\n#     param = {\n#         'objective': 'multi:softprob',\n#         'num_class': len(TARS),\n#         'tree_method': 'gpu_hist',  # use 'gpu_hist' for GPU\n#         'lambda': trial.suggest_loguniform('lambda', 1e-4, 10.0),\n#         'alpha': trial.suggest_loguniform('alpha', 1e-4, 10.0),\n#         'colsample_bytree': trial.suggest_categorical('colsample_bytree', [0.5, 0.6, 0.7, 0.8, 0.9, 1.0]),\n#         'subsample': trial.suggest_categorical('subsample', [0.6, 0.7, 0.8, 0.9, 1.0]),\n#         'learning_rate': trial.suggest_categorical('learning_rate', [0.008, 0.01, 0.02, 0.05, 0.1]),\n#         'n_estimators': 1000,\n#         'max_depth': trial.suggest_categorical('max_depth', [5, 7, 9, 11, 13]),\n#         'min_child_weight': trial.suggest_int('min_child_weight', 1, 300),\n#     }\n\n#     gkf = GroupKFold(n_splits=5)\n#     cv_scores = []\n\n#     for train_index, valid_index in gkf.split(train, train.target, train.patient_id):\n#         X_train, X_valid = train.loc[train_index, FEATURES], train.loc[valid_index, FEATURES]\n#         y_train, y_valid = train.loc[train_index, 'target'].map(TARS), train.loc[valid_index, 'target'].map(TARS)\n\n#         model = xgb.XGBClassifier(**param)\n#         model.fit(X_train, y_train, eval_set=[(X_valid, y_valid)], verbose=False, early_stopping_rounds=10)\n#         preds = model.predict_proba(X_valid)\n#         cv_scores.append(log_loss(y_valid, preds))\n\n#     return np.mean(cv_scores)\n\n# study = optuna.create_study(direction='minimize')\n# study.optimize(objective, n_trials=10)  # Increase n_trials for more extensive search\n\n# print('Number of finished trials:', len(study.trials))\n# print('Best trial:', study.best_trial.params)","metadata":{"execution":{"iopub.status.busy":"2025-11-04T19:59:10.948418Z","iopub.execute_input":"2025-11-04T19:59:10.948801Z","iopub.status.idle":"2025-11-04T19:59:10.956911Z","shell.execute_reply.started":"2025-11-04T19:59:10.948761Z","shell.execute_reply":"2025-11-04T19:59:10.955316Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"* [I 2024-01-15 06:06:55,585] A new study created in memory with name: no-name-21c5e987-7941-4f05-8e92-2903b9a7e304\n* [I 2024-01-15 06:20:36,414] Trial 0 finished with value: 1.0734795198486586 and parameters: {'lambda': 8.578460041884545, 'alpha': 1.0420327876364774, 'colsample_bytree': 1.0, 'subsample': 1.0, 'learning_rate': 0.02, 'max_depth': 7, 'min_child_weight': 30}. Best is trial 0 with value: 1.0734795198486586.\n* [I 2024-01-15 06:39:35,894] Trial 1 finished with value: 1.0792154540612213 and parameters: {'lambda': 0.0029055528938438085, 'alpha': 0.03244985452963714, 'colsample_bytree': 0.5, 'subsample': 1.0, 'learning_rate': 0.01, 'max_depth': 11, 'min_child_weight': 77}. Best is trial 0 with value: 1.0734795198486586.\n* [I 2024-01-15 06:42:19,332] Trial 2 finished with value: 1.107279543251576 and parameters: {'lambda': 1.5919274718287213, 'alpha': 0.042136459342788604, 'colsample_bytree': 0.7, 'subsample': 0.7, 'learning_rate': 0.1, 'max_depth': 5, 'min_child_weight': 152}. Best is trial 0 with value: 1.0734795198486586.\n* [I 2024-01-15 06:57:56,577] Trial 3 finished with value: 1.107956784549986 and parameters: {'lambda': 6.873357111288177, 'alpha': 0.05538406375764404, 'colsample_bytree': 0.5, 'subsample': 0.8, 'learning_rate': 0.008, 'max_depth': 7, 'min_child_weight': 108}. Best is trial 0 with value: 1.0734795198486586.\n* [I 2024-01-15 07:12:36,921] Trial 4 finished with value: 1.1453593669298952 and parameters: {'lambda': 0.0012348624625841455, 'alpha': 6.698933350539058, 'colsample_bytree': 0.8, 'subsample': 0.9, 'learning_rate': 0.01, 'max_depth': 9, 'min_child_weight': 258}. Best is trial 0 with value: 1.0734795198486586.\n* [I 2024-01-15 07:27:45,543] Trial 5 finished with value: 1.1234561427497631 and parameters: {'lambda': 0.0779496745099949, 'alpha': 0.001997110519034328, 'colsample_bytree': 0.5, 'subsample': 0.7, 'learning_rate': 0.008, 'max_depth': 11, 'min_child_weight': 145}. Best is trial 0 with value: 1.0734795198486586.\n* [I 2024-01-15 07:30:45,104] Trial 6 finished with value: 1.139466297500727 and parameters: {'lambda': 0.011643281906929, 'alpha': 0.06334100511005662, 'colsample_bytree': 0.6, 'subsample': 0.7, 'learning_rate': 0.1, 'max_depth': 9, 'min_child_weight': 274}. Best is trial 0 with value: 1.0734795198486586.\n* [I 2024-01-15 07:33:47,478] Trial 7 finished with value: 1.074140292040741 and parameters: {'lambda': 5.487810272015954, 'alpha': 2.266845998962579, 'colsample_bytree': 0.6, 'subsample': 0.8, 'learning_rate': 0.1, 'max_depth': 7, 'min_child_weight': 51}. Best is trial 0 with value: 1.0734795198486586.\n* [I 2024-01-15 07:48:19,863] Trial 8 finished with value: 1.1705891200455967 and parameters: {'lambda': 0.03333036846711228, 'alpha': 0.0004482362892025373, 'colsample_bytree': 0.9, 'subsample': 0.8, 'learning_rate': 0.008, 'max_depth': 13, 'min_child_weight': 294}. Best is trial 0 with value: 1.0734795198486586.\n* [I 2024-01-15 07:53:41,442] Trial 9 finished with value: 1.115599617296683 and parameters: {'lambda': 0.000284724944614318, 'alpha': 0.020480207664040264, 'colsample_bytree': 0.6, 'subsample': 0.9, 'learning_rate': 0.05, 'max_depth': 9, 'min_child_weight': 248}. Best is trial 0 with value: 1.0734795198486586.\nNumber of finished trials: 10\n\n* **Best trial: {'lambda': 8.578460041884545, 'alpha': 1.0420327876364774, 'colsample_bytree': 1.0, 'subsample': 1.0, 'learning_rate': 0.02, 'max_depth': 7, 'min_child_weight': 30}**","metadata":{}},{"cell_type":"code","source":"TOP = 30\n\n# Assuming 'model' is your trained model\nfeature_importance = model.feature_importances_\n\n# Get the feature names from 'train'\nfeature_names = train.columns\n\n# Sort the feature importances and get the indices of the sorted array\nsorted_idx = np.argsort(feature_importance)\n\n# Plot only the top 'TOP' features\nfig = plt.figure(figsize=(10, 8))\nplt.barh(np.arange(len(sorted_idx))[-TOP:], feature_importance[sorted_idx][-TOP:], align='center')\nplt.yticks(np.arange(len(sorted_idx))[-TOP:], feature_names[sorted_idx][-TOP:])\nplt.title(f'Feature Importance - Top {TOP}')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:56:46.674367Z","iopub.execute_input":"2025-11-04T18:56:46.674893Z","iopub.status.idle":"2025-11-04T18:56:47.188945Z","shell.execute_reply.started":"2025-11-04T18:56:46.674853Z","shell.execute_reply":"2025-11-04T18:56:47.187172Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:56:47.191201Z","iopub.execute_input":"2025-11-04T18:56:47.191666Z","iopub.status.idle":"2025-11-04T18:56:47.211328Z","shell.execute_reply.started":"2025-11-04T18:56:47.191630Z","shell.execute_reply":"2025-11-04T18:56:47.209433Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# PATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\n# spec = pd.read_parquet(f'{PATH2}{s}.parquet')\n# spec","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:56:47.212641Z","iopub.execute_input":"2025-11-04T18:56:47.212969Z","iopub.status.idle":"2025-11-04T18:56:47.218491Z","shell.execute_reply.started":"2025-11-04T18:56:47.212945Z","shell.execute_reply":"2025-11-04T18:56:47.217118Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# %%time\n# # READ ALL TEST SPECTROGRAMS\n# PATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\n# files = os.listdir(PATH2)\n# print(f'There are {len(files)} spectrogram parquets')\n\n# spectrograms = {}\n# for i,f in enumerate(files):\n#     if i%100==0: print(i,', ',end='')\n#     tmp = pd.read_parquet(f'{PATH2}{f}')\n#     name = int(f.split('.')[0])\n#     spectrograms_test[name] = tmp.iloc[:,1:].values\n","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:56:47.220478Z","iopub.execute_input":"2025-11-04T18:56:47.220800Z","iopub.status.idle":"2025-11-04T18:56:47.236732Z","shell.execute_reply.started":"2025-11-04T18:56:47.220776Z","shell.execute_reply":"2025-11-04T18:56:47.235072Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# %time\n# # ENGINEER FEATURES\n# import warnings\n# warnings.filterwarnings('ignore')\n\n# # The code generates features from the spectrogram data for use in a model \n# # The features are derived by calculating the mean and minimum values over time for each of the 400 spectrogram frequencies.\n# # Two types of windows are used for these calculations:\n# # A 10-minute window (_mean_10m, _min_10m).\n# # A 20-second window (_mean_20s, _min_20s).\n# # This process results in 1600 features (400 features × 4 calculations) for each EEG ID.\n\n# SPEC_COLS = pd.read_parquet(f'{PATH}1000086677.parquet').columns[1:]\n# FEATURES = [f'{c}_mean_10m' for c in SPEC_COLS]\n# FEATURES += [f'{c}_min_10m' for c in SPEC_COLS]\n# FEATURES += [f'{c}_mean_20s' for c in SPEC_COLS]\n# FEATURES += [f'{c}_min_20s' for c in SPEC_COLS]\n# print(f'We are creating {len(FEATURES)} features for {len(test)} rows... ',end='')\n\n\n# # A data matrix data is initialized to store the new features for each eeg_id in the train DataFrame.\n# # For each row in train, the code calculates the mean and minimum values within the specified 10-minute and 20-second windows.\n# # These calculated values are then stored in the data matrix.\n# # Finally, the matrix is added to the train DataFrame as new columns.\n\n# data = np.zeros((len(test),len(FEATURES)))\n# for k in range(len(test)):\n#     if k%100==0: print(k,', ',end='')\n#     row = test.iloc[k]\n            \n#     # 10 MINUTE WINDOW FEATURES\n#     x = np.nanmean( spec.iloc[:,1:].values, axis=0)\n#     data[k,:400] = x\n#     x = np.nanmin( spec.iloc[:,1:].values, axis=0)\n#     data[k,400:800] = x\n\n#     # 20 SECOND WINDOW FEATURES\n#     x = np.nanmean( spec.iloc[145:155,1:].values, axis=0)\n#     data[k,800:1200] = x\n#     x = np.nanmin( spec.iloc[145:155,1:].values, axis=0)\n#     data[k,1200:1600] = x\n\n#     test[FEATURES] = data\n\n    \n# print()\n# print('New test shape:',test.shape)","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:56:47.238775Z","iopub.execute_input":"2025-11-04T18:56:47.239254Z","iopub.status.idle":"2025-11-04T18:56:47.255649Z","shell.execute_reply.started":"2025-11-04T18:56:47.239229Z","shell.execute_reply":"2025-11-04T18:56:47.253329Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.impute import SimpleImputer\n\n# # Initialize a PCA model\n# pca = PCA(n_components=0.95)\n# print(\"PCA model initialized.\")\n\n# # Initialize an array for original features\n# num_rows = len(test)\n# num_features = 20 * n_channels  # 20 features per channel\n# data_original = np.zeros((num_rows, num_features))\n\n# print(\"Starting feature extraction and PCA processing...\")\n# start_time = time.time()\n\n# for k in range(num_rows):\n#     if k % 1000 == 0:\n#         print(f\"Processing row {k} of {num_rows}...\")\n\n#     row = train.iloc[k]\n#     eeg_segment = spectrograms_test[853520][r:r+300, :]\n\n#     # Apply the feature extraction function to each EEG channel\n#     all_channel_features = []\n#     for i in range(n_channels):\n#         channel_features = extract_frequency_band_features(eeg_segment[:, i])\n#         all_channel_features.extend(channel_features)\n    \n#     data_original[k, :] = all_channel_features\n\n# print(\"Data matrix constructed\")\n\n# # Impute NaN values in the data matrix\n# imputer = SimpleImputer(strategy='mean')\n# data_imputed = imputer.fit_transform(data_original)\n\n# print(f\"NaN values handled. Imputed data matrix shape: {data_imputed.shape}\")\n\n# # Apply PCA on the imputed data\n# pca.fit(data_imputed)\n# print(\"PCA fitting completed.\")\n\n# # Transform data using PCA\n# data_pca = pca.transform(data_imputed)\n\n# # Add PCA features to DataFrame\n# pca_feature_columns = [f'pca_feature_{i}' for i in range(data_pca.shape[1])]\n# test[pca_feature_columns] = data_pca\n\n# # Measure total processing time\n# total_time = time.time() - start_time\n# print(f\"Total processing time: {total_time:.2f} seconds.\")\n\n# test.head()","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:56:47.260749Z","iopub.execute_input":"2025-11-04T18:56:47.261779Z","iopub.status.idle":"2025-11-04T18:56:47.282042Z","shell.execute_reply.started":"2025-11-04T18:56:47.261703Z","shell.execute_reply":"2025-11-04T18:56:47.280964Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Columns to be excluded from scaling\n# excluded_columns = ['eeg_id', 'spectrogram_id', 'patient_id']\n\n# # Save the columns to be excluded\n# excluded_data = test[excluded_columns]\n\n# # DataFrame with only the columns to be scaled\n# features = test.drop(columns=excluded_columns)\n\n# # Initialize the StandardScaler\n# scaler = StandardScaler()\n\n# # Fit the scaler to the features and transform them\n# features_scaled = scaler.fit_transform(features)\n\n# # Create a DataFrame from the scaled features\n# features_scaled_df = pd.DataFrame(features_scaled, columns=features.columns)\n\n# # Concatenate the scaled features with the excluded columns\n# test_scaled_df = pd.concat([excluded_data.reset_index(drop=True),features_scaled_df,], axis=1)\n# test_scaled_df \n","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:56:47.284461Z","iopub.execute_input":"2025-11-04T18:56:47.285578Z","iopub.status.idle":"2025-11-04T18:56:47.306267Z","shell.execute_reply.started":"2025-11-04T18:56:47.285540Z","shell.execute_reply":"2025-11-04T18:56:47.304964Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# FEATURE ENGINEER TEST\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\ndata = np.zeros((len(test),len(FEATURES)))\n    \nfor k in range(len(test)):\n    row = test.iloc[k]\n    s = int( row.spectrogram_id )\n    spec = pd.read_parquet(f'{PATH2}{s}.parquet')\n    \n    # 10 MINUTE WINDOW FEATURES\n    x = np.nanmean( spec.iloc[:,1:].values, axis=0)\n    data[k,:400] = x\n    x = np.nanmin( spec.iloc[:,1:].values, axis=0)\n    data[k,400:800] = x\n\n    # 20 SECOND WINDOW FEATURES\n    x = np.nanmean( spec.iloc[145:155,1:].values, axis=0)\n    data[k,800:1200] = x\n    x = np.nanmin( spec.iloc[145:155,1:].values, axis=0)\n    data[k,1200:1600] = x\n\ntest[FEATURES] = data\nprint('New test shape',test.shape)","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:56:47.307657Z","iopub.execute_input":"2025-11-04T18:56:47.307962Z","iopub.status.idle":"2025-11-04T18:56:47.789796Z","shell.execute_reply.started":"2025-11-04T18:56:47.307933Z","shell.execute_reply":"2025-11-04T18:56:47.788177Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# INFER XGBOOST ON TEST\npreds = []\n\nfor i in range(5):\n    print(i, ', ', end='')\n    \n    # Load the XGBoost model\n    model = xgb.XGBClassifier()\n    model.load_model(f'XGB_v{VER}_f{i}.model')\n    \n    # Make predictions\n    pred = model.predict_proba(test[FEATURES])\n    preds.append(pred)\n\n# Average the predictions from each fold\npred = np.mean(preds, axis=0)\nprint()\nprint('Test preds shape', pred.shape)","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:56:47.792000Z","iopub.execute_input":"2025-11-04T18:56:47.792440Z","iopub.status.idle":"2025-11-04T18:56:48.673930Z","shell.execute_reply.started":"2025-11-04T18:56:47.792406Z","shell.execute_reply":"2025-11-04T18:56:48.672368Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CÉLULA 23.1A — Diagnóstico de confiança das predições no TEST (sem rótulos)\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nprobs_max = np.max(pred, axis=1)\nentropy = -np.sum(np.clip(pred, 1e-12, 1).astype(float) * np.log(np.clip(pred, 1e-12, 1).astype(float)), axis=1)\n\nprint(f'Confiança média (max prob): {probs_max.mean():.4f} ± {probs_max.std():.4f}')\nprint(f'Entropia média: {entropy.mean():.4f} ± {entropy.std():.4f}')\n\n# distribuição da classe predita\nclasses = ['Seizure','LPD','GPD','LRDA','GRDA','Other']\npred_labels = np.argmax(pred, axis=1)\nclass_counts = pd.Series(pred_labels).value_counts().reindex(range(len(classes)), fill_value=0)\ndisp = pd.DataFrame({'class': classes, 'count': class_counts.values, 'pct': class_counts.values / len(pred)})\nprint('\\nDistribuição de classes previstas no TEST:')\nprint(disp)\n\n# histograma de confiança\nplt.figure(figsize=(6,4))\nplt.hist(probs_max, bins=20, edgecolor='k')\nplt.title('Histograma da confiança (max prob) — TEST')\nplt.xlabel('max prob'); plt.ylabel('freq')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-04T20:25:08.958128Z","iopub.execute_input":"2025-11-04T20:25:08.958554Z","iopub.status.idle":"2025-11-04T20:25:09.194688Z","shell.execute_reply.started":"2025-11-04T20:25:08.958522Z","shell.execute_reply":"2025-11-04T20:25:09.193499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[TARGETS] = pred\nsub.to_csv('submission.csv',index=False)\nprint('Submission shape',sub.shape)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:56:48.675537Z","iopub.execute_input":"2025-11-04T18:56:48.675900Z","iopub.status.idle":"2025-11-04T18:56:48.705683Z","shell.execute_reply.started":"2025-11-04T18:56:48.675871Z","shell.execute_reply":"2025-11-04T18:56:48.703353Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# SANITY CHECK TO CONFIRM PREDICTIONS SUM TO ONE\nsub.iloc[:,-6:].sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2025-11-04T18:56:48.707780Z","iopub.execute_input":"2025-11-04T18:56:48.708275Z","iopub.status.idle":"2025-11-04T18:56:48.721941Z","shell.execute_reply.started":"2025-11-04T18:56:48.708245Z","shell.execute_reply":"2025-11-04T18:56:48.720189Z"},"trusted":true},"outputs":[],"execution_count":null}]}