{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7392775,"sourceType":"datasetVersion","datasetId":4297782},{"sourceId":7447509,"sourceType":"datasetVersion","datasetId":4334995}],"dockerImageVersionId":30648,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, gc, random, warnings, time, sklearn\nimport pandas as pd, numpy as np, matplotlib.pyplot as plt\nimport fastai\nfrom fastai.tabular.all import *\nfrom sklearn.model_selection import train_test_split\nfrom fastai.vision.all import *\n\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:13:28.971405Z","iopub.execute_input":"2024-02-05T23:13:28.971785Z","iopub.status.idle":"2024-02-05T23:13:40.43107Z","shell.execute_reply.started":"2024-02-05T23:13:28.971758Z","shell.execute_reply":"2024-02-05T23:13:40.429868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"laptop=False\nif laptop:\n    train_path = 'data/hms-harmful-brain-activity-classification/train.csv'\n    test_path = 'data/hms-harmful-brain-activity-classification/test.csv'\n   \n    kaggle_multiple_spectrograms_path = 'data/hms-harmful-brain-activity-classification/train_spectrograms'\n    kaggle_single_spectrograms_path = 'data/brain-spectrograms/specs.npy'\n   \n    eeg_multiple_spectrograms_path = 'data/hms-harmful-brain-activity-classification/train_eegs/'\n    eeg_single_spectrograms_path = 'data/brain-eeg-spectrograms/eeg_specs.npy'\n\n    test_spectrograms_path = 'data/hms-harmful-brain-activity-classification/test_spectrograms'\n    test_eegs_path = 'data/hms-harmful-brain-activity-classification/test_eegs'\nelse:\n    train_path = '/kaggle/input/hms-harmful-brain-activity-classification/train.csv'\n    test_path = '/kaggle/input/hms-harmful-brain-activity-classification/test.csv'\n    \n    kaggle_multiple_spectrograms_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\n    kaggle_single_spectrograms_path = '/kaggle/input/brain-spectrograms/specs.npy'\n\n    eeg_multiple_spectrograms_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs'\n    eeg_single_spectrograms_path = '/kaggle/input/brain-eeg-spectrograms/eeg_specs.npy'\n\n\ndef submit_simple_prediction():\n    try: pd.__version__\n    except NameError as e: import pandas as pd\n    \n    test = pd.read_csv(test_path)\n    targets = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n    \n    preds = []\n    for _ in range(len(test)):\n        preds.append([1/6 for _ in range(6)])\n    \n    submission = pd.DataFrame({'eeg_id': test.eeg_id.values})\n    submission[targets] = preds\n    submission.to_csv('submission.csv', index=False)\n    return test, submission\n\n# test, submission = submit_simple_prediction()\n# test.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-05T23:13:40.433331Z","iopub.execute_input":"2024-02-05T23:13:40.433767Z","iopub.status.idle":"2024-02-05T23:13:40.44494Z","shell.execute_reply.started":"2024-02-05T23:13:40.43373Z","shell.execute_reply":"2024-02-05T23:13:40.444022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(train_path)\ntargets = df.columns[-6:]\nprint(targets)","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:13:40.446081Z","iopub.execute_input":"2024-02-05T23:13:40.446845Z","iopub.status.idle":"2024-02-05T23:13:40.713911Z","shell.execute_reply.started":"2024-02-05T23:13:40.44682Z","shell.execute_reply":"2024-02-05T23:13:40.712691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(df, targets):\n    agg_dict = {'spectrogram_id': 'first', 'spectrogram_label_offset_seconds': 'min'}\n\n    train = df.groupby('eeg_id')[['spectrogram_id', 'spectrogram_label_offset_seconds']].agg(agg_dict)\n    train.columns = ['spec_id', 'min_offset']\n\n    max_offset = df.groupby('eeg_id')['spectrogram_label_offset_seconds'].max()\n    train['max_offset'] = max_offset\n\n    patient_id = df.groupby('eeg_id')['patient_id'].first()\n    train['patiend_id'] = patient_id\n\n    target_sums = df.groupby('eeg_id')[targets].sum()\n    for target in targets:\n        train[target] = target_sums[target].values\n\n    y_data = train[targets].values\n    y_data_normalised = y_data/y_data.sum(axis=1, keepdims=True)\n    train[targets] = y_data_normalised\n\n    consensus = df.groupby('eeg_id')['expert_consensus'].first()\n    train['targets'] = consensus\n\n    train = train.reset_index()\n    return train\n\ntrain = preprocess(df, targets)\nlen(train)","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:13:40.716657Z","iopub.execute_input":"2024-02-05T23:13:40.717126Z","iopub.status.idle":"2024-02-05T23:13:40.80764Z","shell.execute_reply.started":"2024-02-05T23:13:40.717087Z","shell.execute_reply":"2024-02-05T23:13:40.806515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def load_kaggle_spectrograms(read_individual_files=False):\n#     t0 = time.time()\n\n#     if read_individual_files:\n#         files = os.listdir(kaggle_multiple_spectrograms_path)\n#         spectrograms = {}\n#         for index, file in enumerate(files):\n#             if index % 1000 == 0:\n#                 print((f'index = {index};' + \n#                        f'time elapsed = {(time.time() - t0):.2f} s'))\n#             temp = pd.read_parquet(f'{kaggle_multiple_spectrograms_path}/{file}')\n#             name = int(file.split('.')[0])\n#             spectrograms[name] = temp.iloc[:,1:].values\n#     else:\n#         spectrograms = np.load(kaggle_single_spectrograms_path, allow_pickle=True).item()\n#         print(f'time to load all kaggle spectrograms from specs.npy: {time.time()-t0:.2f}s')\n#     return spectrograms\n                \n# kaggle_spectrograms = load_kaggle_spectrograms(read_individual_files=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:13:40.809219Z","iopub.execute_input":"2024-02-05T23:13:40.809575Z","iopub.status.idle":"2024-02-05T23:13:40.81494Z","shell.execute_reply.started":"2024-02-05T23:13:40.809545Z","shell.execute_reply":"2024-02-05T23:13:40.813902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_eeg_spectrograms(read_individual_files=False):\n    t0 = time.time()\n    if read_individual_files:\n        eeg_spectrograms = {}\n        for index, eeg_id in enumerate(train.eeg_id.values):\n            if index%100==0: print(f'index = {index}; time elapsed = {time.time()-t0:.2f}s')\n\n            eeg_spectrogram = np.load(f'{eeg_multiple_spectrograms_path}{eeg_id}.parquet', allow_pickle=True)\n            eeg_spectrograms[eeg_id] = eeg_spectrogram\n    else:\n        eeg_spectrograms = np.load(eeg_single_spectrograms_path, allow_pickle=True).item()\n        print(f'time to load all eeg spectrograms from specs.npy: {time.time()-t0:.2f}s')\n    return eeg_spectrograms\n\neeg_spectrograms = load_eeg_spectrograms(read_individual_files=False)","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:13:40.816731Z","iopub.execute_input":"2024-02-05T23:13:40.817064Z","iopub.status.idle":"2024-02-05T23:14:51.349036Z","shell.execute_reply.started":"2024-02-05T23:13:40.817034Z","shell.execute_reply":"2024-02-05T23:14:51.347837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# eeg_id = train.iloc[0].eeg_id\n# # plt.imshow(eeg_spectrograms[eeg_id])\n# x = np.zeros((512, 256), dtype='float32')\n\n# eeg_spectrogram = eeg_spectrograms[eeg_id]\n# for j in range(4):\n#     x[j*128:(j+1)*128] = eeg_spectrogram[:,:,j]","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:14:51.3503Z","iopub.execute_input":"2024-02-05T23:14:51.350671Z","iopub.status.idle":"2024-02-05T23:14:51.355702Z","shell.execute_reply.started":"2024-02-05T23:14:51.350638Z","shell.execute_reply":"2024-02-05T23:14:51.354502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# len(eeg_spectrograms)","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:14:51.357187Z","iopub.execute_input":"2024-02-05T23:14:51.357521Z","iopub.status.idle":"2024-02-05T23:14:51.367827Z","shell.execute_reply.started":"2024-02-05T23:14:51.357489Z","shell.execute_reply":"2024-02-05T23:14:51.366879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = train.iloc[:,5:11]\n\nfrom sklearn.model_selection import train_test_split\ntrain_idx, test_idx = train_test_split(list(range(len(train))), test_size=0.2)\ntrain_idx, val_idx = train_test_split(list(range(len(train))), test_size=0.2)\n\ndef get_x(i):\n    eeg_id = train.iloc[i].eeg_id\n    x = np.zeros((512, 256), dtype='float32')\n    eeg_spectrogram = eeg_spectrograms[eeg_id]\n    for j in range(4):\n        x[j*128:(j+1)*128] = eeg_spectrogram[:,:,j]\n    return x\n\ndef get_y(i):\n    return labels.iloc[i]","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:14:51.369214Z","iopub.execute_input":"2024-02-05T23:14:51.369526Z","iopub.status.idle":"2024-02-05T23:14:51.397489Z","shell.execute_reply.started":"2024-02-05T23:14:51.369497Z","shell.execute_reply":"2024-02-05T23:14:51.396652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dblock = DataBlock(blocks=(ImageBlock, RegressionBlock(n_out=6)),\n                   get_x=get_x,\n                   get_y=get_y,\n                   splitter=IndexSplitter(val_idx))  # Use IndexSplitter to split by indexes\n\n# # Create DataLoaders\ndls = dblock.dataloaders(range(len(train)), bs=32)\n\n# # Define the model\nlearn = cnn_learner(dls, resnet18, y_range=(0, 1), loss_func=MSELossFlat())\n\n# # Find an appropriate learning rate\n# learn.lr_find()","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:14:51.400409Z","iopub.execute_input":"2024-02-05T23:14:51.400899Z","iopub.status.idle":"2024-02-05T23:14:52.844582Z","shell.execute_reply.started":"2024-02-05T23:14:51.400874Z","shell.execute_reply":"2024-02-05T23:14:52.843457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Train the model\nlearn.fit_one_cycle(1, lr_max=slice(1e-6, 1e-4))","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:14:52.846084Z","iopub.execute_input":"2024-02-05T23:14:52.846492Z","iopub.status.idle":"2024-02-05T23:16:13.036799Z","shell.execute_reply.started":"2024-02-05T23:14:52.846453Z","shell.execute_reply":"2024-02-05T23:16:13.035719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del train; gc.collect()\ntest = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nprint('Test shape',test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:16:13.04004Z","iopub.execute_input":"2024-02-05T23:16:13.040987Z","iopub.status.idle":"2024-02-05T23:16:13.062468Z","shell.execute_reply.started":"2024-02-05T23:16:13.040946Z","shell.execute_reply":"2024-02-05T23:16:13.061562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pywt, librosa\n\nUSE_WAVELET = None \n\nNAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\n\n# DENOISE FUNCTION\ndef maddest(d, axis=None):\n    return np.mean(np.absolute(d - np.mean(d, axis)), axis)\n\ndef denoise(x, wavelet='haar', level=1):    \n    coeff = pywt.wavedec(x, wavelet, mode=\"per\")\n    sigma = (1/0.6745) * maddest(coeff[-level])\n\n    uthresh = sigma * np.sqrt(2*np.log(len(x)))\n    coeff[1:] = (pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:])\n\n    ret=pywt.waverec(coeff, wavelet, mode='per')\n    \n    return ret\n\ndef spectrogram_from_eeg(parquet_path, display=False):\n    \n    # LOAD MIDDLE 50 SECONDS OF EEG SERIES\n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n    \n    # VARIABLE TO HOLD SPECTROGRAM\n    img = np.zeros((128,256,4),dtype='float32')\n    \n    if display: plt.figure(figsize=(10,7))\n    signals = []\n    for k in range(4):\n        COLS = FEATS[k]\n        \n        for kk in range(4):\n        \n            # COMPUTE PAIR DIFFERENCES\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values\n\n            # FILL NANS\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n            else: x[:] = 0\n\n            # DENOISE\n            if USE_WAVELET:\n                x = denoise(x, wavelet=USE_WAVELET)\n            signals.append(x)\n\n            # RAW SPECTROGRAM\n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, \n                  n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128)\n\n            # LOG TRANSFORM\n            width = (mel_spec.shape[1]//32)*32\n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n\n            # STANDARDIZE TO -1 TO 1\n            mel_spec_db = (mel_spec_db+40)/40 \n            img[:,:,k] += mel_spec_db\n                \n        # AVERAGE THE 4 MONTAGE DIFFERENCES\n        img[:,:,k] /= 4.0\n        \n        if display:\n            plt.subplot(2,2,k+1)\n            plt.imshow(img[:,:,k],aspect='auto',origin='lower')\n            plt.title(f'EEG {eeg_id} - Spectrogram {NAMES[k]}')\n            \n    if display: \n        plt.show()\n        plt.figure(figsize=(10,5))\n        offset = 0\n        for k in range(4):\n            if k>0: offset -= signals[3-k].min()\n            plt.plot(range(10_000),signals[k]+offset,label=NAMES[3-k])\n            offset += signals[3-k].max()\n        plt.legend()\n        plt.title(f'EEG {eeg_id} Signals')\n        plt.show()\n        print(); print('#'*25); print()\n        \n    return img\n\n# CREATE ALL EEG SPECTROGRAMS\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\nDISPLAY = 0\nEEG_IDS2 = test.eeg_id.unique()\nall_eegs2 = {}\n\nprint('Converting Test EEG to Spectrograms...'); print()\nfor i,eeg_id in enumerate(EEG_IDS2):\n        \n    # CREATE SPECTROGRAM FROM EEG PARQUET\n    img = spectrogram_from_eeg(f'{PATH2}{eeg_id}.parquet', i<DISPLAY)\n    all_eegs2[eeg_id] = img","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:16:13.064037Z","iopub.execute_input":"2024-02-05T23:16:13.064569Z","iopub.status.idle":"2024-02-05T23:16:23.391595Z","shell.execute_reply.started":"2024-02-05T23:16:13.064532Z","shell.execute_reply":"2024-02-05T23:16:23.390341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import PIL\n\ndef softmax(x):\n    \"\"\"Compute softmax values for each sets of scores in x.\"\"\"\n    e_x = np.exp(x - max(x))\n    return e_x / e_x.sum()  # only difference\n\npreds = []\nfor i in range(len(test)):\n    img = all_eegs2[test.iloc[i].eeg_id]\n    xx = np.zeros((512, 256), dtype='float32')\n    for k in range(4):\n        xx[128*k:128*(k+1)] = img[:,:,k]\n\n    img_pil = PIL.Image.fromarray(xx)\n    \n    a, b, c = learn.predict(img_pil)\n    \n    # Assuming `prediction` is your model's output of 6 floats\n    prediction = np.array(a)  # Convert prediction to a numpy array if it isn't already\n    probabilities = softmax(a)\n    preds.append(probabilities)","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:16:24.886821Z","iopub.execute_input":"2024-02-05T23:16:24.887221Z","iopub.status.idle":"2024-02-05T23:16:25.142327Z","shell.execute_reply.started":"2024-02-05T23:16:24.887184Z","shell.execute_reply":"2024-02-05T23:16:25.141145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame({'eeg_id':test.eeg_id.values})\nsub[targets] = preds\nsub.to_csv('submission.csv',index=False)\nprint('Submissionn shape',sub.shape)\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-05T23:17:02.791402Z","iopub.execute_input":"2024-02-05T23:17:02.791939Z","iopub.status.idle":"2024-02-05T23:17:02.816946Z","shell.execute_reply.started":"2024-02-05T23:17:02.791902Z","shell.execute_reply":"2024-02-05T23:17:02.815153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}