{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nfrom sklearn.preprocessing import StandardScaler\nimport numpy as np\nfrom scipy.io import loadmat\nimport os\nimport dask.dataframe as dd\nfrom dask.multiprocessing import get\nimport warnings\nimport scipy\nimport torch\nfrom tqdm import tqdm\nwarnings.filterwarnings('ignore', category=Warning)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-01T18:35:30.663305Z","iopub.execute_input":"2024-05-01T18:35:30.664077Z","iopub.status.idle":"2024-05-01T18:35:39.595336Z","shell.execute_reply.started":"2024-05-01T18:35:30.664034Z","shell.execute_reply":"2024-05-01T18:35:39.593869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%pip install fcwt\nimport fcwt","metadata":{"execution":{"iopub.status.busy":"2024-05-01T18:35:46.123140Z","iopub.execute_input":"2024-05-01T18:35:46.123579Z","iopub.status.idle":"2024-05-01T18:36:49.538317Z","shell.execute_reply.started":"2024-05-01T18:35:46.123548Z","shell.execute_reply":"2024-05-01T18:36:49.536438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Parameters for CWT\nsampling_freq = 200 #Taken from the dataset, do not change\nlowest_freq = 0.5\nhighest_freq = 40\nnum_freq = 50\nstride = 16\n\nnum_rows = 10000 #hard limit on num rows, some data has more\n\nstart_data_idx = 6000\nend_data_idx = 12000\n\ndef process_eeg_file(filepath, features_eeg):\n    #Generate tensor with premapped feilds\n    num_features = len(features_eeg)\n    \n    raw_CWT_data = np.zeros((end_data_idx - start_data_idx, num_features, num_freq, num_rows // stride), dtype=np.float16)\n\n    data_index = 0\n    for filename in tqdm(list(os.listdir(eeg_data_dir))[start_data_idx:end_data_idx]):\n        if filename.endswith('.parquet'):\n            filepath = os.path.join(eeg_data_dir, filename)\n\n            eeg_data = pd.read_parquet(filepath)\n\n            for col in features_eeg:\n                eeg_data_feat = eeg_data[col]\n                eeg_data_mean = eeg_data_feat.mean()\n                eeg_data_feat.fillna(value=eeg_data_mean, inplace=True)\n\n                signal = eeg_data_feat.to_numpy()[:num_rows]\n                freq, cwt = fcwt.cwt(signal, fs=sampling_freq, f0=lowest_freq, f1=highest_freq, fn=num_freq)\n                strided_cwt = cwt[:, ::stride]\n\n                raw_CWT_data[data_index, features_eeg.index(col), :] = strided_cwt\n            data_index += 1\n    return raw_CWT_data\n\nfeatures_only_eeg = ['Fp1', 'F3', 'C3', 'P3', 'F7', 'T3', 'T5', 'O1', 'Fz', 'Cz', 'Pz', 'Fp2', 'F4', 'C4', 'P4', 'F8', 'T4', 'T6', 'O2']\neeg_data_dir = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\nCWT = process_eeg_file(eeg_data_dir, features_only_eeg)\ndata_file_name = \"CWT_data_\" + str(start_data_idx) + \"_\" + str(end_data_idx)\nnp.save(data_file_name, CWT)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-28T22:05:21.151338Z","iopub.execute_input":"2024-04-28T22:05:21.151853Z","iopub.status.idle":"2024-04-28T22:25:13.514676Z","shell.execute_reply.started":"2024-04-28T22:05:21.151798Z","shell.execute_reply":"2024-04-28T22:25:13.513213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#For pulling the labels\nstart_data_idx = 0\nend_data_idx = 6000\n\nlabel_cols = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\n\ndef process_eeg_file_labels(filepath):\n\n    #Generate tensor with premapped feilds    \n    labels = np.zeros((end_data_idx - start_data_idx, len(label_cols)), dtype=np.float16)\n    train_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\n    data_index = 0\n    for filename in tqdm(list(os.listdir(eeg_data_dir))[start_data_idx:end_data_idx]):\n        if filename.endswith('.parquet'):\n            filepath = os.path.join(eeg_data_dir, filename)\n            dot_index = filename.index('.parquet')\n            eeg_data_id = filename[:dot_index]\n            matching_row_summed = train_df.loc[train_df['eeg_id'] == int(eeg_data_id)][label_cols].sum()\n            mean_label_probability = matching_row_summed / matching_row_summed.sum()\n            labels[data_index, :] = mean_label_probability.to_numpy()\n        data_index += 1\n\neeg_data_dir = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\nlabel_file_name = \"label_data_\" + str(start_data_idx) + \"_\" + str(end_data_idx)\nlabels = process_eeg_file_labels(eeg_data_dir)\nnp.save(label_file_name, labels)","metadata":{"execution":{"iopub.status.busy":"2024-05-01T19:15:58.144978Z","iopub.execute_input":"2024-05-01T19:15:58.145400Z","iopub.status.idle":"2024-05-01T19:16:08.213804Z","shell.execute_reply.started":"2024-05-01T19:15:58.145357Z","shell.execute_reply":"2024-05-01T19:16:08.212613Z"},"trusted":true},"execution_count":null,"outputs":[]}]}