{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":2378330,"sourceType":"datasetVersion","datasetId":492658},{"sourceId":7698353,"sourceType":"datasetVersion","datasetId":4493465}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Based on @cdeotte's and @rafaelzimmermann1's work\nHere is a dataset creator for creating spectrograms with gpu...\nHere are some important caveats to consider before you step forward\n- dataset size is around 40 gb for my current configuration so I divided it into two notebooks\n- you can control temportal and frequency resolution by changing parameters like n_fft, nperseg\n- This computes full and complete spectrograms... assuming that we will need all 100k labels and not some random 10_000 samples \n- The part above is important, to access the specific 10,000 samples around the label you'd need to write additional code (a few lines), depending on your temporal resolution","metadata":{}},{"cell_type":"code","source":"%%time\n# Installation of RAPIDS to Use cuSignal\nimport sys\n\n!cp ../input/rapids/rapids.0.17.0 /opt/conda/envs/rapids.tar.gz\n!cd /opt/conda/envs/ && tar -xzvf rapids.tar.gz > /dev/null\n!rm /opt/conda/envs/rapids.tar.gz\n\nsys.path += [\"/opt/conda/envs/rapids/lib/python3.7/site-packages\"]\nsys.path += [\"/opt/conda/envs/rapids/lib/python3.7\"]\nsys.path += [\"/opt/conda/envs/rapids/lib\"]\n!cp /opt/conda/envs/rapids/lib/libxgboost.so /opt/conda/lib/","metadata":{"execution":{"iopub.status.busy":"2024-02-25T17:21:17.031839Z","iopub.execute_input":"2024-02-25T17:21:17.032545Z","iopub.status.idle":"2024-02-25T17:22:39.452685Z","shell.execute_reply.started":"2024-02-25T17:21:17.032515Z","shell.execute_reply":"2024-02-25T17:22:39.451513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport matplotlib.pyplot as plt\nimport gc\nimport json\nimport pickle\nimport torch\nfrom tqdm import tqdm\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-25T17:22:39.455685Z","iopub.execute_input":"2024-02-25T17:22:39.456068Z","iopub.status.idle":"2024-02-25T17:22:45.683291Z","shell.execute_reply.started":"2024-02-25T17:22:39.456029Z","shell.execute_reply":"2024-02-25T17:22:45.682169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DEVICE = 'cuda' if torch.cuda.is_available() else 'cpu'","metadata":{"execution":{"iopub.status.busy":"2024-02-25T17:22:45.684537Z","iopub.execute_input":"2024-02-25T17:22:45.684967Z","iopub.status.idle":"2024-02-25T17:22:45.740044Z","shell.execute_reply.started":"2024-02-25T17:22:45.684939Z","shell.execute_reply":"2024-02-25T17:22:45.738940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CONFIG = {\n    \"TEMPORAL_RESOLUTION\" : 256,\n    \"WINDOW_LENGTH\" : 128,\n    \"FREQUENCY_BANDS\" : 128,\n    \"F_MIN\": 0,\n    \"F_MAX\" : 20,\n    \"N_FFT\" : 1024,\n    'WAVELET' : 'haar',\n    'DENOISE_LEVEL':1\n}","metadata":{"execution":{"iopub.status.busy":"2024-02-25T17:22:45.743025Z","iopub.execute_input":"2024-02-25T17:22:45.743388Z","iopub.status.idle":"2024-02-25T17:22:48.974338Z","shell.execute_reply.started":"2024-02-25T17:22:45.743358Z","shell.execute_reply":"2024-02-25T17:22:48.973421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\neeg_lengths_df = pd.read_csv(\"/kaggle/input/hms-hba-eeg-to-length/eeg_to_length.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-25T17:22:48.975822Z","iopub.execute_input":"2024-02-25T17:22:48.976435Z","iopub.status.idle":"2024-02-25T17:22:49.262138Z","shell.execute_reply.started":"2024-02-25T17:22:48.976402Z","shell.execute_reply":"2024-02-25T17:22:49.261262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NAMES = ['LL','LP','RP','RR']\n\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\n\ndirectory_path = 'EEG_Spectrograms/'\nif not os.path.exists(directory_path):\n    os.makedirs(directory_path)","metadata":{"execution":{"iopub.status.busy":"2024-02-25T17:22:49.263318Z","iopub.execute_input":"2024-02-25T17:22:49.263668Z","iopub.status.idle":"2024-02-25T17:22:49.270101Z","shell.execute_reply.started":"2024-02-25T17:22:49.263605Z","shell.execute_reply":"2024-02-25T17:22:49.269134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# DENOISING (via wavelets)\n ","metadata":{}},{"cell_type":"code","source":"import pywt\nprint(pywt.wavelist())","metadata":{"execution":{"iopub.status.busy":"2024-02-25T17:22:49.271247Z","iopub.execute_input":"2024-02-25T17:22:49.271632Z","iopub.status.idle":"2024-02-25T17:22:49.493041Z","shell.execute_reply.started":"2024-02-25T17:22:49.271582Z","shell.execute_reply":"2024-02-25T17:22:49.491886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"USE_WAVELET = None #or \"db8\" or anything below\n","metadata":{"execution":{"iopub.status.busy":"2024-02-25T17:22:49.494263Z","iopub.execute_input":"2024-02-25T17:22:49.494791Z","iopub.status.idle":"2024-02-25T17:22:49.499717Z","shell.execute_reply.started":"2024-02-25T17:22:49.494758Z","shell.execute_reply":"2024-02-25T17:22:49.498543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DENOISE FUNCTION\ndef maddest(d, axis=None):\n    return np.mean(np.absolute(d - np.mean(d, axis)), axis)\n\ndef denoise(x, wavelet='haar', level=1):    \n    coeff = pywt.wavedec(x, wavelet, mode=\"per\")\n    sigma = (1/0.6745) * maddest(coeff[-level])\n\n    uthresh = sigma * np.sqrt(2*np.log(len(x)))\n    coeff[1:] = (pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:])\n\n    ret=pywt.waverec(coeff, wavelet, mode='per')\n    \n    return ret","metadata":{"execution":{"iopub.status.busy":"2024-02-25T17:22:49.500847Z","iopub.execute_input":"2024-02-25T17:22:49.501151Z","iopub.status.idle":"2024-02-25T17:22:49.513328Z","shell.execute_reply.started":"2024-02-25T17:22:49.501126Z","shell.execute_reply":"2024-02-25T17:22:49.512584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa","metadata":{"execution":{"iopub.status.busy":"2024-02-25T17:22:49.516664Z","iopub.execute_input":"2024-02-25T17:22:49.517452Z","iopub.status.idle":"2024-02-25T17:22:49.529970Z","shell.execute_reply.started":"2024-02-25T17:22:49.517426Z","shell.execute_reply":"2024-02-25T17:22:49.529185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's explain these variables.\n\n- y is the input time series signal\n- sr is the sampling frequency. In this competition EEG is sample 200 times per sec\n- hop_length produces image with width = len(x)/hop_length (determins resolution (smaller = higher temportal resolution))\n- n_fft controls vertical resolution and quality of spectrogram\n- n_mels produces image with height = n_mels\n- fmin is smallest frequency in our spectrogram\n- fmax is largest frequency in our spectrogram\n- win_length controls hortizonal resolution and quality of spectrogram","metadata":{}},{"cell_type":"code","source":"def spectrogram_from_eeg(parquet_path, display=False, config = None):\n#    load 50 seconds \n    if config is None:\n        TEMPORAL_RESOLUTION = 256\n        WINDOW_LENGTH = 128\n        FREQUENCY_BANDS = 128\n        F_MIN = 0\n        F_MAX = 20\n        N_FFT = 1024\n        WAVELET = 'haar'\n        DENOISE_LEVEL=1\n    else:\n        TEMPORAL_RESOLUTION = config['TEMPORAL_RESOLUTION']\n        WINDOW_LENGTH = config['WINDOW_LENGTH']\n        FREQUENCY_BANDS = config['FREQUENCY_BANDS']\n        F_MIN = config['F_MIN']\n        F_MAX = config['F_MAX']\n        N_FFT = config['N_FFT']\n        WAVELET = config['WAVELET']\n        DENOISE_LEVEL = config['DENOISE_LEVEL']\n    \n    eeg = pd.read_parquet(parquet_path)\n    middle = (len(eeg)-10_000)//2\n    eeg = eeg.iloc[middle:middle+10_000]\n    \n    \n#     variable to hold spectrogram\n# 128 X 256 (freq X time...  ) (btw time can be controlled here by hop length)\n    img = np.zeros((FREQUENCY_BANDS,TEMPORAL_RESOLUTION,4),dtype='float32')\n    if display: plt.figure(figsize=(12,7))\n        \n        \n    signals = []\n    \n    for k in range(4):\n#         4 chains\n        COLS = FEATS[k]\n    \n        for kk in range(4):\n            x = eeg[COLS[kk]].values - eeg[COLS[kk+1]].values # temporal chain signal composites... (k1 - k2)\n\n#             replace nans with nanmean (mean without nans) (this way mean will still be the same)\n            m = np.nanmean(x)\n            if np.isnan(x).mean()<1 :\n                x = np.nan_to_num(x, nan=m)\n            else:\n#                 if all are nans, replace with zeros\n                x[:] = 0\n#             denoise\n            if WAVELET:\n                x = denoise(x, wavelet=WAVELET, level=DENOISE_LEVEL)\n            signals.append(x)\n            \n            mel_spec = librosa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//TEMPORAL_RESOLUTION, \n                  n_fft=N_FFT, n_mels=FREQUENCY_BANDS, fmin=F_MIN, fmax=F_MAX, win_length=WINDOW_LENGTH)\n\n\n\n            \n            \n            width = (mel_spec.shape[1]//32)*32\n            \n            time_steps = librosa.times_like(mel_spec, sr=200, hop_length=len(x) // TEMPORAL_RESOLUTION)[:width]\n            \n            mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max).astype(np.float32)[:,:width]\n#             standardize \n            mel_spec_db = (mel_spec_db+40)/40 \n            img[:,:,k] += mel_spec_db\n            \n            img[:,:,k] /= 4.0\n            \n            if display:\n                plt.subplot(2,2,k+1)\n                plt.imshow(img[:,:,k],aspect='auto',origin='lower')\n                plt.title(f'EEG {eeg_id} - Spectrogram {NAMES[k]}')\n\n    if display: \n        plt.show()\n        plt.figure(figsize=(10,5))\n        offset = 0\n        for k in range(4):\n            if k>0: offset -= signals[3-k].min()\n            plt.plot(range(10_000),signals[k]+offset,label=NAMES[3-k])\n            offset += signals[3-k].max()\n        plt.legend()\n        plt.title(f'EEG {eeg_id} Signals')\n        plt.show()\n        print(); print('#'*25); print()\n\n    return img, time_steps","metadata":{"execution":{"iopub.status.busy":"2024-02-25T17:22:49.531239Z","iopub.execute_input":"2024-02-25T17:22:49.531513Z","iopub.status.idle":"2024-02-25T17:22:49.548970Z","shell.execute_reply.started":"2024-02-25T17:22:49.531490Z","shell.execute_reply":"2024-02-25T17:22:49.548155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cupy as cp\nimport cusignal\nimport numpy as np\nfrom scipy.ndimage import gaussian_filter\nfrom scipy.signal import butter, filtfilt, iirnotch\n\ndef create_spectrogram_with_cusignal(eeg_data, eeg_id, start):\n    electrode_names = ['LL', 'RL', 'LP', 'RP']\n\n    electrode_pairs = [\n        ['Fp1', 'F7', 'T3', 'T5', 'O1'],\n        ['Fp2', 'F8', 'T4', 'T6', 'O2'],\n        ['Fp1', 'F3', 'C3', 'P3', 'O1'],\n        ['Fp2', 'F4', 'C4', 'P4', 'O2']\n    ]\n    \n    # Filter specifications\n    nyquist_freq = 0.5 * 200\n    low_cut_freq_normalized = 1.8 / nyquist_freq\n    high_cut_freq_normalized = 20 / nyquist_freq\n\n    # Bandpass and notch filter\n    bandpass_coefficients = butter(5, [low_cut_freq_normalized, high_cut_freq_normalized], btype='band')\n    notch_coefficients = iirnotch(w0=60, Q=30, fs=200)\n\n    spec_size =  len(eeg_data) # 10_000\n    start = start * 0\n    eeg_data = eeg_data.iloc[start:start+spec_size]\n\n    # Initialize spectrogram container\n    \n#     spectrogram = cp.zeros((128, 256, 4), dtype='float32')\n    spectrogram = None\n    \n    processed_eeg = {}\n\n    for i, name in enumerate(electrode_names):\n        cols = electrode_pairs[i]\n        processed_eeg[name] = np.zeros(spec_size)\n        for j in range(4):\n            # Compute differential signals\n            signal = cp.array(eeg_data[cols[j]].values - eeg_data[cols[j+1]].values)\n\n            # Handle NaNs\n            mean_signal = cp.nanmean(signal)\n            signal = cp.nan_to_num(signal, nan=mean_signal) if cp.isnan(signal).mean() < 1 else cp.zeros_like(signal)\n            \n\n            # Filter bandpass and notch\n            signal_filtered = filtfilt(*notch_coefficients, signal.get())\n            signal_filtered = filtfilt(*bandpass_coefficients, signal_filtered)\n            signal = cp.asarray(signal_filtered)\n                \n            # Spectrogram parameters\n            fs = 200  \n            nperseg = 48\n            noverlap = 9\n            nfft = 1024 +248\n            \n            \n            # GPU-accelerated spectrogram computation\n            frequencies, times, Sxx = cusignal.spectrogram(signal, fs, nperseg=nperseg, noverlap=noverlap, nfft=nfft)\n\n            # Filter frequency range\n            valid_freqs = (frequencies >= 0.59) & (frequencies <= 20)\n            frequencies_filtered = frequencies[valid_freqs]\n            Sxx_filtered = Sxx[valid_freqs, :]\n\n            # Logarithmic transformation and normalization using Cupy\n            spectrogram_slice = cp.clip(Sxx_filtered, cp.exp(-4), cp.exp(6))\n            spectrogram_slice = cp.log10(spectrogram_slice)\n\n            normalization_epsilon = 1e-6\n            mean = spectrogram_slice.mean(axis=(0, 1), keepdims=True)\n            std = spectrogram_slice.std(axis=(0, 1), keepdims=True)\n            spectrogram_slice = (spectrogram_slice - mean) / (std + normalization_epsilon)\n            \n            if spectrogram is  None:\n                spectrogram = spectrogram_slice[:]\n                spectrogram = np.repeat(spectrogram[:,:,np.newaxis], 4, 2)\n            else:\n                spectrogram[:, :, i] += spectrogram_slice\n            processed_eeg[f'{cols[j]}_{cols[j+1]}'] = signal.get()\n            processed_eeg[name] += signal.get()\n        \n    # Convert to NumPy and apply Gaussian filter\n    spectrogram_np = cp.asnumpy(spectrogram)\n    spectrogram_np = gaussian_filter(spectrogram_np, sigma=0.7)\n\n    # Filter EKG signal\n    ekg_signal_filtered = filtfilt(*notch_coefficients, eeg_data[\"EKG\"].values)\n    ekg_signal_filtered = filtfilt(*bandpass_coefficients, ekg_signal_filtered)\n    processed_eeg['EKG'] = np.array(ekg_signal_filtered)\n\n    return spectrogram_np, processed_eeg","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:40:44.026648Z","iopub.execute_input":"2024-02-25T15:40:44.026958Z","iopub.status.idle":"2024-02-25T15:40:49.758974Z","shell.execute_reply.started":"2024-02-25T15:40:44.026929Z","shell.execute_reply":"2024-02-25T15:40:49.758157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# FREQ_COLUMNS = [f\"{name} {f}\" for name in NAMES for f in librosa.core.mel_frequencies(n_mels=CONFIG[\"FREQUENCY_BANDS\"], fmin=CONFIG[\"F_MIN\"], fmax=CONFIG[\"F_MAX\"])]","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:40:49.760180Z","iopub.execute_input":"2024-02-25T15:40:49.760856Z","iopub.status.idle":"2024-02-25T15:40:49.765135Z","shell.execute_reply.started":"2024-02-25T15:40:49.760822Z","shell.execute_reply":"2024-02-25T15:40:49.764224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def save_numpy_to_parquet(img, path, time_steps):\n#     n_time = img.shape[1]\n    \n#     df = pd.DataFrame(img.reshape(n_time, -1), columns=FREQ_COLUMNS)\n#     df['time'] = time_steps\n    \n#     df.to_parquet(path)\n# #     to df\n    \n            \n# # save parquet","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:40:49.768197Z","iopub.execute_input":"2024-02-25T15:40:49.768507Z","iopub.status.idle":"2024-02-25T15:40:49.778528Z","shell.execute_reply.started":"2024-02-25T15:40:49.768483Z","shell.execute_reply":"2024-02-25T15:40:49.777824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EEG_IDS = df.eeg_id.unique()","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:40:49.779465Z","iopub.execute_input":"2024-02-25T15:40:49.779731Z","iopub.status.idle":"2024-02-25T15:40:49.803307Z","shell.execute_reply.started":"2024-02-25T15:40:49.779709Z","shell.execute_reply":"2024-02-25T15:40:49.802406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# eeg_to_spec = {}\n# for _,row in df[df['eeg_id'].isin(EEG_IDS)].drop_duplicates(subset='eeg_id')[[\"eeg_id\",\"spectrogram_id\"]].iterrows():\n    \n#     eeg_to_spec[row['eeg_id']] = row['spectrogram_id']","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:40:49.804448Z","iopub.execute_input":"2024-02-25T15:40:49.804707Z","iopub.status.idle":"2024-02-25T15:40:49.810870Z","shell.execute_reply.started":"2024-02-25T15:40:49.804686Z","shell.execute_reply":"2024-02-25T15:40:49.810180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted_eegs = eeg_lengths_df.sort_values(by='length')\nsorted_eegs = sorted_eegs.reset_index()\nSWAP_INDEX = sorted_eegs[(sorted_eegs['length'] > 40000) & (sorted_eegs['length'] <= 50000)].index[0]","metadata":{"execution":{"iopub.status.busy":"2024-02-25T15:40:49.812049Z","iopub.execute_input":"2024-02-25T15:40:49.812765Z","iopub.status.idle":"2024-02-25T15:40:49.840055Z","shell.execute_reply.started":"2024-02-25T15:40:49.812733Z","shell.execute_reply":"2024-02-25T15:40:49.839397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_size = sorted_eegs.iloc[:]['length'].sum()","metadata":{"execution":{"iopub.status.busy":"2024-02-25T16:02:30.176695Z","iopub.execute_input":"2024-02-25T16:02:30.177126Z","iopub.status.idle":"2024-02-25T16:02:30.182917Z","shell.execute_reply.started":"2024-02-25T16:02:30.177096Z","shell.execute_reply":"2024-02-25T16:02:30.181800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"full_size","metadata":{"execution":{"iopub.status.busy":"2024-02-25T16:02:32.746873Z","iopub.execute_input":"2024-02-25T16:02:32.747940Z","iopub.status.idle":"2024-02-25T16:02:32.753348Z","shell.execute_reply.started":"2024-02-25T16:02:32.747891Z","shell.execute_reply":"2024-02-25T16:02:32.752427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\nDISPLAY = 0\nEEG_IDS = df.eeg_id.unique()\nall_specs = {}\neeg_to_chunk = {}\nL = len(EEG_IDS)//2\nprint(\"Half A n. of EEGs \", L)\n# L=10\n\ncurr_chunk_size = 0 # n.rows\npartition_ratio = 0.34\ncurr_chunk = 1\nfor i in tqdm(range(L)):\n    \n    eeg_id =  sorted_eegs.iloc[i]['eeg_id']        \n        \n    eeg_data = pd.read_parquet(f\"{PATH}{eeg_id}.parquet\")\n    \n    curr_chunk_size += len(eeg_data)\n    \n    if curr_chunk_size/full_size > partition_ratio:\n        curr_chunk_size = 0\n        \n        print(\"chunk filled🌝 Dumping...\")\n        f = open(f\"all_specs_a{curr_chunk}.pkl\",\"wb\")\n        pickle.dump(all_specs, f)\n        \n        f.close()\n        \n        curr_chunk+=1\n        \n        del all_specs\n        gc.collect()\n        all_specs = {} \n    \n    \n    img, _ = create_spectrogram_with_cusignal(eeg_data, None, 0)\n    \n    all_specs[eeg_id] = img\n    eeg_to_chunk[eeg_id] = f\"a{curr_chunk}\"\n    \n    del img\n    del _\n    del eeg_data\n    \nif all_specs != {}:\n    f = open(f\"all_specs_a{curr_chunk}.pkl\",\"wb\")\n    pickle.dump(all_specs, f)\n    f.close()\n    \nf = open(\"eeg_to_chunk_a.pkl\", 'wb')\npickle.dump(eeg_to_chunk, f)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-25T17:29:20.247718Z","iopub.execute_input":"2024-02-25T17:29:20.248094Z","iopub.status.idle":"2024-02-25T17:29:20.550668Z","shell.execute_reply.started":"2024-02-25T17:29:20.248066Z","shell.execute_reply":"2024-02-25T17:29:20.549449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# TEST SPEED OF MEL_SPEC AND GPU_SPEC","metadata":{}},{"cell_type":"markdown","source":"# GPU Specs via cuML","metadata":{}},{"cell_type":"code","source":"# fig, axes =  plt.subplots(2,2) \n# fig.set_figheight(6)\n# fig.set_figwidth(9)\n# axes[0][0].imshow(spec[0][:,:,0], cmap='jet')\n# axes[0][1].imshow(spec[0][:,:,1], cmap='jet')\n# axes[1][0].imshow(spec[0][:,:,2], cmap='jet')\n# axes[1][1].imshow(spec[0][:,:,3], cmap='jet')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-25T12:04:38.842338Z","iopub.execute_input":"2024-02-25T12:04:38.842739Z","iopub.status.idle":"2024-02-25T12:04:39.451330Z","shell.execute_reply.started":"2024-02-25T12:04:38.842699Z","shell.execute_reply":"2024-02-25T12:04:39.450356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cpu_specs =spectrogram_from_eeg(eegs[10], display=False, config = None)\n# fig, axes =  plt.subplots(2,2) \n# fig.set_figheight(6)\n# fig.set_figwidth(9)\n# axes[0][0].imshow(cpu_specs[0][:,:,0], cmap='jet')\n# axes[0][1].imshow(cpu_specs[0][:,:,1], cmap='jet')\n# axes[1][0].imshow(cpu_specs[0][:,:,2], cmap='jet')\n# axes[1][1].imshow(cpu_specs[0][:,:,3], cmap='jet')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-25T12:04:40.994908Z","iopub.execute_input":"2024-02-25T12:04:40.995814Z","iopub.status.idle":"2024-02-25T12:04:41.909962Z","shell.execute_reply.started":"2024-02-25T12:04:40.995781Z","shell.execute_reply":"2024-02-25T12:04:41.909087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Moment of truth\nGPU vs CPU for specs","metadata":{"execution":{"iopub.status.busy":"2024-02-25T10:13:49.512154Z","iopub.execute_input":"2024-02-25T10:13:49.512549Z","iopub.status.idle":"2024-02-25T10:13:49.547637Z","shell.execute_reply.started":"2024-02-25T10:13:49.512516Z","shell.execute_reply":"2024-02-25T10:13:49.546343Z"}}},{"cell_type":"code","source":"# import cupy as cp\n# import cusignal\n# import numpy as np\n# from scipy.ndimage import gaussian_filter\n# from scipy.signal import butter, filtfilt, iirnotch\n\n# def create_spectrogram_with_cusignal(eeg_data, eeg_id, start):\n#     electrode_names = ['LL', 'RL', 'LP', 'RP']\n\n#     electrode_pairs = [\n#         ['Fp1', 'F7', 'T3', 'T5', 'O1'],\n#         ['Fp2', 'F8', 'T4', 'T6', 'O2'],\n#         ['Fp1', 'F3', 'C3', 'P3', 'O1'],\n#         ['Fp2', 'F4', 'C4', 'P4', 'O2']\n#     ]\n    \n#     # Filter specifications\n#     nyquist_freq = 0.5 * 200\n#     low_cut_freq_normalized = 1.8 / nyquist_freq\n#     high_cut_freq_normalized = 20 / nyquist_freq\n\n#     # Bandpass and notch filter\n#     bandpass_coefficients = butter(5, [low_cut_freq_normalized, high_cut_freq_normalized], btype='band')\n#     notch_coefficients = iirnotch(w0=60, Q=30, fs=200)\n\n#     spec_size = 10_000\n#     start = start * 200\n#     eeg_data = eeg_data.iloc[start:start+spec_size]\n\n#     # Initialize spectrogram container\n    \n#     spectrogram = cp.zeros((128, 256, 4), dtype='float32')\n    \n#     processed_eeg = {}\n\n#     for i, name in enumerate(electrode_names):\n#         cols = electrode_pairs[i]\n#         processed_eeg[name] = np.zeros(spec_size)\n#         for j in range(4):\n#             # Compute differential signals\n#             signal = cp.array(eeg_data[cols[j]].values - eeg_data[cols[j+1]].values)\n\n#             # Handle NaNs\n#             mean_signal = cp.nanmean(signal)\n#             signal = cp.nan_to_num(signal, nan=mean_signal) if cp.isnan(signal).mean() < 1 else cp.zeros_like(signal)\n            \n\n#             # Filter bandpass and notch\n#             signal_filtered = filtfilt(*notch_coefficients, signal.get())\n#             signal_filtered = filtfilt(*bandpass_coefficients, signal_filtered)\n#             signal = cp.asarray(signal_filtered)\n                \n#             # Spectrogram parameters\n#             fs = 200  \n#             nperseg = 48\n#             noverlap = 9\n#             nfft = 1024+290\n            \n            \n#             # GPU-accelerated spectrogram computation\n#             frequencies, times, Sxx = cusignal.spectrogram(signal, fs, nperseg=nperseg, noverlap=noverlap, nfft=nfft)\n\n#             # Filter frequency range\n#             valid_freqs = (frequencies >= 0.) & (frequencies <= 20)\n#             frequencies_filtered = frequencies[valid_freqs]\n#             Sxx_filtered = Sxx[valid_freqs, :]\n\n#             # Logarithmic transformation and normalization using Cupy\n#             spectrogram_slice = cp.clip(Sxx_filtered, cp.exp(-4), cp.exp(6))\n#             spectrogram_slice = cp.log10(spectrogram_slice)\n\n#             normalization_epsilon = 1e-6\n#             mean = spectrogram_slice.mean(axis=(0, 1), keepdims=True)\n#             std = spectrogram_slice.std(axis=(0, 1), keepdims=True)\n#             spectrogram_slice = (spectrogram_slice - mean) / (std + normalization_epsilon)\n            \n#             spectrogram[:, :, i] += spectrogram_slice[:spectrogram.shape[0], :spectrogram.shape[1]]\n#             processed_eeg[f'{cols[j]}_{cols[j+1]}'] = signal.get()\n#             processed_eeg[name] += signal.get()\n\n#     # Convert to NumPy and apply Gaussian filter\n#     spectrogram_np = cp.asnumpy(spectrogram)\n#     spectrogram_np = gaussian_filter(spectrogram_np, sigma=0.7)\n\n#     # Filter EKG signal\n#     ekg_signal_filtered = filtfilt(*notch_coefficients, eeg_data[\"EKG\"].values)\n#     ekg_signal_filtered = filtfilt(*bandpass_coefficients, ekg_signal_filtered)\n#     processed_eeg['EKG'] = np.array(ekg_signal_filtered)\n\n#     return spectrogram_np, processed_eeg","metadata":{"execution":{"iopub.status.busy":"2024-02-25T11:39:43.231053Z","iopub.execute_input":"2024-02-25T11:39:43.231425Z","iopub.status.idle":"2024-02-25T11:39:43.238543Z","shell.execute_reply.started":"2024-02-25T11:39:43.231397Z","shell.execute_reply":"2024-02-25T11:39:43.237475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# eeg_lengths_df = pd.read_csv(\"/kaggle/input/hms-hba-eeg-to-length/eeg_to_length.csv\")\n# df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-25T14:50:07.900942Z","iopub.execute_input":"2024-02-25T14:50:07.901297Z","iopub.status.idle":"2024-02-25T14:50:08.076407Z","shell.execute_reply.started":"2024-02-25T14:50:07.901267Z","shell.execute_reply":"2024-02-25T14:50:08.075424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df[df['eeg_id'].isin(eeg_lengths_df.iloc[:16500]['eeg_id'])]","metadata":{"execution":{"iopub.status.busy":"2024-02-25T14:50:37.459300Z","iopub.execute_input":"2024-02-25T14:50:37.459640Z","iopub.status.idle":"2024-02-25T14:50:37.485160Z","shell.execute_reply.started":"2024-02-25T14:50:37.459614Z","shell.execute_reply":"2024-02-25T14:50:37.483693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df)","metadata":{"execution":{"iopub.status.busy":"2024-02-25T14:50:41.855268Z","iopub.execute_input":"2024-02-25T14:50:41.855641Z","iopub.status.idle":"2024-02-25T14:50:41.863237Z","shell.execute_reply.started":"2024-02-25T14:50:41.855614Z","shell.execute_reply":"2024-02-25T14:50:41.861137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}