{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"f03def11-0e2a-411b-b7fb-8933976a1363","cell_type":"code","source":"!pip install imbalanced-learn==0.11.0 --quiet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T18:24:07.229954Z","iopub.execute_input":"2025-10-30T18:24:07.230255Z","iopub.status.idle":"2025-10-30T18:24:12.881068Z","shell.execute_reply.started":"2025-10-30T18:24:07.230231Z","shell.execute_reply":"2025-10-30T18:24:12.879819Z"}},"outputs":[],"execution_count":null},{"id":"b8fb8d0d","cell_type":"code","source":"import pandas as pd, numpy as np, os\nimport matplotlib.pyplot as plt, gc\n\ntrain = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nprint('Train shape', train.shape )\ndisplay(train.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T18:24:12.883221Z","iopub.execute_input":"2025-10-30T18:24:12.883615Z","iopub.status.idle":"2025-10-30T18:24:13.543267Z","shell.execute_reply.started":"2025-10-30T18:24:12.883576Z","shell.execute_reply":"2025-10-30T18:24:13.542465Z"}},"outputs":[],"execution_count":null},{"id":"5a1715fc","cell_type":"code","source":"NOMES = ['LL','LP','RP','RR'] #CADEIA TEMPORAL ESQUERDA(CTE), CADEIA PARASAGITAL ESQUERDA(CPE), CADEIA PARASAGITAL DIREITA(CPD), CADEIA TEMPORAL DIREITA(CTD)\n\nELETRODOS = [['Fp1','F7','T3','T5','O1'],\n            ['Fp1','F3','C3','P3','O1'],\n            ['Fp2','F8','T4','T6','O2'],\n            ['Fp2','F4','C4','P4','O2']]\n\ndirectory_path = 'EEG_Spectrograms/'\nif not os.path.exists(directory_path):\n    os.makedirs(directory_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T18:24:13.544133Z","iopub.execute_input":"2025-10-30T18:24:13.544443Z","iopub.status.idle":"2025-10-30T18:24:13.550258Z","shell.execute_reply.started":"2025-10-30T18:24:13.544391Z","shell.execute_reply":"2025-10-30T18:24:13.549451Z"}},"outputs":[],"execution_count":null},{"id":"9e2fff40","cell_type":"code","source":"import pywt\nprint(\"The wavelet functions we can use:\")\nprint(pywt.wavelist())\n\nUSE_WAVELET = 'db8' #or \"db8\" or anything below","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T18:24:13.551209Z","iopub.execute_input":"2025-10-30T18:24:13.551532Z","iopub.status.idle":"2025-10-30T18:24:13.8529Z","shell.execute_reply.started":"2025-10-30T18:24:13.551505Z","shell.execute_reply":"2025-10-30T18:24:13.851956Z"}},"outputs":[],"execution_count":null},{"id":"0d7a5b7a","cell_type":"code","source":"def maddest(d, axis=None):\n    return np.mean(np.absolute(d - np.mean(d, axis)), axis)\n\ndef denoise(x, wavelet='haar', level=1):    \n    coeff = pywt.wavedec(x, wavelet, mode=\"per\")\n    sigma = (1 / 0.6745) * maddest(coeff[-level])\n    \n    uthresh = sigma * np.sqrt(2 * np.log(len(x)))\n    \n    # CORREÇÃO AQUI ↓↓↓↓↓↓↓↓↓↓↓↓↓↓↓↓↓↓↓↓↓↓↓↓↓↓↓\n    coeff[1:] = [pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:]]\n\n    ret = pywt.waverec(coeff, wavelet, mode='per')\n    \n    return ret\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T18:24:13.855175Z","iopub.execute_input":"2025-10-30T18:24:13.855601Z","iopub.status.idle":"2025-10-30T18:24:13.862438Z","shell.execute_reply.started":"2025-10-30T18:24:13.855578Z","shell.execute_reply":"2025-10-30T18:24:13.861311Z"}},"outputs":[],"execution_count":null},{"id":"e76a4d12","cell_type":"code","source":"import librosa\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\ndef eeg_espectrograma(parquet_path, display=False, use_wavelet=False):\n    \n    # CARREGA 50 SEGUNDOS DO EEG\n    eeg = pd.read_parquet(parquet_path)\n    meio = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[meio:meio+10_000]\n    \n    # VARIAVEL PARA SEGURAR O ESPECTROGRAMA\n    img = np.zeros((128,256,4),dtype='float32')\n    \n    if display: plt.figure(figsize=(10,7))\n    sinais = []\n    for k in range(4):\n        COLS = ELETRODOS[k]\n        \n        for kk in range(4):\n        \n            # DIFERENÇA DE PARES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # PREENCHE RUIDO/NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1:\n                x = np.nan_to_num(x, nan=m)\n            else:\n                x[:] = 0\n\n            # REMOVE RUIDO SE USAR WAVELET\n            if use_wavelet:\n                \n                x = denoise(x, wavelet=USE_WAVELET)  # ou outra sua função de wavelet\n            \n            sinais.append(x)\n\n            # RAW SPECTROGRAM\n            mel_espec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                  n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_espec.shape[1]//32)*32\n            mel_spec_db = librosa.power_to_db(mel_espec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            mel_spec_db = (mel_spec_db+40)/40 \n            img[:,:,k] += mel_spec_db\n                \n        # MEDIA DAS 4 DIFERENÇAS\n        img[:,:,k] /= 4.0\n        \n        if display:\n            plt.subplot(2,2,k+1)\n            plt.imshow(img[:,:,k],aspect='auto',origin='lower')\n            plt.title(f'EEG {eeg_id} - espectrograma {NOMES[k]}')\n            \n    if display: \n        plt.show()\n        plt.figure(figsize=(10,5))\n        offset = 0\n        for k in range(4):\n            if k>0: offset -= sinais[3-k].min()\n            plt.plot(range(10_000),sinais[k]+offset,label=NOMES[3-k])\n            offset += sinais[3-k].max()\n        plt.legend()\n        plt.title(f'EEG {eeg_id} sinais')\n        plt.show()\n        print(); print('#'*25); print()\n        \n    return img\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T18:24:13.863663Z","iopub.execute_input":"2025-10-30T18:24:13.864001Z","iopub.status.idle":"2025-10-30T18:24:13.888902Z","shell.execute_reply.started":"2025-10-30T18:24:13.863972Z","shell.execute_reply":"2025-10-30T18:24:13.887938Z"}},"outputs":[],"execution_count":null},{"id":"03041939-3e56-4e17-92d5-c4a9bbf2a4f6","cell_type":"code","source":"%%time\nPATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\nDISPLAY = 4\nEEG_IDS = train.eeg_id.unique()\nall_eegs = {}\n\nfor i,eeg_id in enumerate(EEG_IDS):\n    if (i%100==0)&(i!=0): print(i,', ',end='')\n        \n    # CREATE SPECTROGRAM FROM EEG PARQUET\n    img = eeg_espectrograma(f'{PATH}{eeg_id}.parquet', i<DISPLAY, use_wavelet=True)\n    \n    # SAVE TO DISK\n    if i==DISPLAY:\n        print(f'Creating and writing {len(EEG_IDS)} spectrograms to disk... ',end='')\n    np.save(f'{directory_path}{eeg_id}',img)\n    all_eegs[eeg_id] = img\n   \n# SAVE EEG SPECTROGRAM DICTIONARY\nnp.save('eeg_specs',all_eegs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T18:24:13.890016Z","iopub.execute_input":"2025-10-30T18:24:13.890337Z","iopub.status.idle":"2025-10-30T19:16:30.978067Z","shell.execute_reply.started":"2025-10-30T18:24:13.890308Z","shell.execute_reply":"2025-10-30T19:16:30.974594Z"}},"outputs":[],"execution_count":null},{"id":"fd60d81d-b96f-4b05-8ad6-97bdbffee5a0","cell_type":"code","source":"from collections import Counter\n\n# rótulo majoritário por eeg_id\nlabel_por_eeg = (\n    train.groupby('eeg_id')['expert_consensus']\n         .agg(lambda s: Counter(s).most_common(1)[0][0])\n)\n\nprint(label_por_eeg.head())\nprint('Total EEGs rotulados:', len(label_por_eeg))\nfrom collections import Counter\n\nlabel_por_eeg = (\n    train.groupby('eeg_id')['expert_consensus']\n         .agg(lambda s: Counter(s).most_common(1)[0][0])\n)\n\nprint(label_por_eeg.head())\nprint('Total EEGs rotulados:', len(label_por_eeg))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T19:16:30.981793Z","iopub.execute_input":"2025-10-30T19:16:30.988669Z","iopub.status.idle":"2025-10-30T19:16:31.747648Z","shell.execute_reply.started":"2025-10-30T19:16:30.988609Z","shell.execute_reply":"2025-10-30T19:16:31.74666Z"}},"outputs":[],"execution_count":null},{"id":"b1f167bb-0338-45d5-8389-2a8036830c4e","cell_type":"code","source":"import numpy as np\n\nfrom scipy.stats import skew, kurtosis\n\nimport numpy as np\nfrom scipy.stats import skew, kurtosis\n\ndef spec_to_features(img):\n    \"\"\"\n    img: np.array shape (128, 256, 4)\n    retorna: vetor 1D de features mais ricas (4096)\n    \"\"\"\n    feats = []\n    for k in range(img.shape[2]):  # 4 canais\n        M = img[:, :, k]  # (128 bandas, 256 frames)\n\n        # estatísticas básicas\n        feats.append(M.mean(axis=1))       # 128\n        feats.append(M.std(axis=1))        # 128\n\n        # energia por banda\n        feats.append(np.sum(M**2, axis=1)) # 128\n\n        # skewness e kurtosis com tratamento\n        row_std = M.std(axis=1)\n        skew_vals = np.zeros_like(row_std)\n        kurt_vals = np.zeros_like(row_std)\n\n        mask = row_std > 1e-6  # só calcula se tem variação suficiente\n        skew_vals[mask] = skew(M[mask, :], axis=1)\n        kurt_vals[mask] = kurtosis(M[mask, :], axis=1)\n\n        feats.append(skew_vals)   # 128\n        feats.append(kurt_vals)   # 128\n\n        # percentis por banda\n        feats.append(np.percentile(M, 25, axis=1))  # 128\n        feats.append(np.percentile(M, 50, axis=1))  # 128 (mediana)\n        feats.append(np.percentile(M, 75, axis=1))  # 128\n\n    # concatena tudo em 1D\n    return np.concatenate(feats)\n\n\nfrom sklearn.preprocessing import LabelEncoder\n\nX_list, y_list, ids_ok = [], [], []\n\n# Se você tem tudo em memória:\n# for eeg_id, img in all_eegs.items():\n\n# Se preferir carregar do disco:\n# import os\n# for eeg_id in label_por_eeg.index:\n#     caminho = os.path.join('EEG_Spectrograms', f'{eeg_id}.npy')\n#     if not os.path.exists(caminho): continue\n#     img = np.load(caminho)\n\nfor eeg_id, img in all_eegs.items():\n    if eeg_id in label_por_eeg.index:\n        X_list.append(spec_to_features(img))\n        y_list.append(label_por_eeg.loc[eeg_id])\n        ids_ok.append(eeg_id)\n\nX = np.vstack(X_list).astype('float32')\nle = LabelEncoder()\ny = le.fit_transform(y_list)\n\nprint('X shape:', X.shape)  # (n_eegs, 1024)\nprint('Classes:', list(le.classes_))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T19:16:31.749136Z","iopub.execute_input":"2025-10-30T19:16:31.749506Z","iopub.status.idle":"2025-10-30T19:22:44.221212Z","shell.execute_reply.started":"2025-10-30T19:16:31.749476Z","shell.execute_reply":"2025-10-30T19:22:44.220268Z"}},"outputs":[],"execution_count":null},{"id":"9d42305e-9877-4965-991e-f3f86213c587","cell_type":"code","source":"from sklearn.model_selection import GroupShuffleSplit\n\n# mapeia eeg_id -> patient_id\npaciente_por_eeg = train.groupby('eeg_id')['patient_id'].first()\ngroups = np.array([paciente_por_eeg.loc[e] for e in ids_ok])\n\ngss = GroupShuffleSplit(test_size=0.2, random_state=42)\ntr_idx, te_idx = next(gss.split(X, y, groups=groups))\n\nX_tr, X_te = X[tr_idx], X[te_idx]\ny_tr, y_te = y[tr_idx], y[te_idx]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T19:22:44.222457Z","iopub.execute_input":"2025-10-30T19:22:44.222701Z","iopub.status.idle":"2025-10-30T19:22:44.570429Z","shell.execute_reply.started":"2025-10-30T19:22:44.222679Z","shell.execute_reply":"2025-10-30T19:22:44.569506Z"}},"outputs":[],"execution_count":null},{"id":"a669fdef-b693-4212-a4b4-402f12407b2c","cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\nX_tr = scaler.fit_transform(X_tr)\nX_te = scaler.transform(X_te)\nprint(\"Normalização concluída.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T19:22:44.571521Z","iopub.execute_input":"2025-10-30T19:22:44.57177Z","iopub.status.idle":"2025-10-30T19:22:45.9563Z","shell.execute_reply.started":"2025-10-30T19:22:44.571752Z","shell.execute_reply":"2025-10-30T19:22:45.955473Z"}},"outputs":[],"execution_count":null},{"id":"a37c322d-7fa4-424c-8c61-fe97991ef7b2","cell_type":"code","source":"# Instalação e imports necessários\nimport os, random, numpy as np, tensorflow as tf\nfrom imblearn.over_sampling import SMOTE\nfrom sklearn.utils import class_weight\n\nSEED = 42\nos.environ['PYTHONHASHSEED'] = str(SEED)\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\nprint(\"TensorFlow version:\", tf.__version__)\n\n# Balanceamento opcional (SMOTE)\nUSE_SMOTE = True\nif USE_SMOTE:\n    sm = SMOTE(random_state=SEED, n_jobs=-1)\n    X_tr_res, y_tr_res = sm.fit_resample(X_tr, y_tr)\n    print(\"Antes SMOTE:\", X_tr.shape, \"Depois SMOTE:\", X_tr_res.shape)\nelse:\n    X_tr_res, y_tr_res = X_tr, y_tr\n\n# Class weights para rede neural\ncw = class_weight.compute_class_weight('balanced', classes=np.unique(y_tr_res), y=y_tr_res)\nclass_weights = {i: w for i, w in enumerate(cw)}\nprint(\"Class weights:\", class_weights)\n\nNUM_CLASSES = len(le.classes_)\nINPUT_SHAPE = X_tr_res.shape[1]\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"2efc7e3c-a053-43d3-b685-0107a7e5e9a8","cell_type":"code","source":"from tensorflow.keras import layers, models, callbacks, optimizers\n\ndef build_mlp(input_dim, num_classes, dropout=0.35, hidden_units=[1024, 512, 256]):\n    inp = layers.Input(shape=(input_dim,))\n    x = inp\n    for units in hidden_units:\n        x = layers.Dense(units)(x)\n        x = layers.BatchNormalization()(x)\n        x = layers.Activation('relu')(x)\n        x = layers.Dropout(dropout)(x)\n    out = layers.Dense(num_classes, activation='softmax')(x)\n    return models.Model(inp, out)\n\nmodel = build_mlp(INPUT_SHAPE, NUM_CLASSES)\nmodel.compile(optimizer=optimizers.Adam(learning_rate=1e-3),\n              loss='sparse_categorical_crossentropy',\n              metrics=['accuracy'])\nmodel.summary()\n\nes = callbacks.EarlyStopping(monitor='val_loss', patience=8, restore_best_weights=True)\nrl = callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=4, min_lr=1e-6)\n\nhistory = model.fit(\n    X_tr_res, y_tr_res,\n    validation_split=0.15,\n    epochs=100,\n    batch_size=64,\n    class_weight=class_weights,\n    callbacks=[es, rl],\n    verbose=2\n)\n\nmodel.save('eeg_mlp_model.h5')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-30T19:23:07.332726Z","iopub.execute_input":"2025-10-30T19:23:07.333739Z"}},"outputs":[],"execution_count":null},{"id":"d5f4d1d5-506e-4fad-b551-12c2b074df27","cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix, ConfusionMatrixDisplay, accuracy_score\nimport matplotlib.pyplot as plt\n\ny_proba = model.predict(X_te, batch_size=128)\ny_pred = np.argmax(y_proba, axis=1)\n\nprint(\"Accuracy (NN):\", accuracy_score(y_te, y_pred))\nprint(classification_report(y_te, y_pred, target_names=le.classes_, digits=4))\n\ncm = confusion_matrix(y_te, y_pred, labels=np.arange(NUM_CLASSES))\ndisp = ConfusionMatrixDisplay(cm, display_labels=le.classes_)\nfig, ax = plt.subplots(figsize=(8,6))\ndisp.plot(ax=ax, xticks_rotation=45, cmap='Blues', values_format='d')\nax.set_title('NN – Matriz de Confusão')\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"id":"9d2d9090-bb6c-4991-a9a8-ce2b04376553","cell_type":"code","source":"# Substitua seu bloco atual por este:\nimport numpy as np\nimp = rf.feature_importances_\nprint('Soma importâncias:', imp.sum())\n\nby_chain = []\nlabels = ['LL', 'LP', 'RP', 'RR']\n\nfor k in range(4):\n    start = k * (len(imp) // 4)  # divisão automática\n    end = start + (len(imp) // 4)\n    by_chain.append(imp[start:end].sum())\n\n# Loop CORRIGIDO:\nfor k, val in enumerate(by_chain):\n    print(f'Cadeia {k} ({labels[k]}): {val:.4f}')","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}