{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport torch\nimport pandas as pd\nfrom tqdm import tqdm\nfrom scipy.stats import entropy\nfrom scipy.signal import butter, lfilter, freqz\nimport os\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-10T01:37:42.033305Z","iopub.execute_input":"2024-04-10T01:37:42.033938Z","iopub.status.idle":"2024-04-10T01:37:42.040277Z","shell.execute_reply.started":"2024-04-10T01:37:42.033904Z","shell.execute_reply":"2024-04-10T01:37:42.039037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = torch.tensor([1,2,3,4,5,6,7,8,9,10])\nprint(x.shape)\nx = x.view(5,2)\nx = x.permute(1,0)\nprint(x.shape)\nprint(x)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T01:37:42.042640Z","iopub.execute_input":"2024-04-10T01:37:42.043128Z","iopub.status.idle":"2024-04-10T01:37:42.059235Z","shell.execute_reply.started":"2024-04-10T01:37:42.043085Z","shell.execute_reply":"2024-04-10T01:37:42.057811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#PythonでPyTorchを使用したサンプルコードを書いてみます。このコードでは、初めに仮想のEEGデータを作成し、その後に説明した形状変更とチャネルの追加処理を行います。最初のステップとして、バッチサイズbs、チャネル数16、時間軸10000のデータを生成し、それを画像データとして扱うための変換を実施します。\nimport torch\n\n# バッチサイズ、チャネル数、時間軸の長さを設定\nbs = 64  # 例えば、バッチサイズを64とする\nchannels = 16\ntime_length = 10000  # 時間軸の長さ\n\n# 仮想のEEGデータを生成 (形状: bs x channels x time_length)\nx = torch.randn(bs, channels, time_length)\nprint(x.shape)\n\n# データを2D画像のように扱うために次元の拡張を行う\n# 実際には、ここでのコメントに対応するコードが省略されていますが、\n# x = x.unsqueeze(1)  # 形状を (bs, 1, 16, 10000) に変更\n\n# 形状変更と次元の入れ替え\nreshaped_tensor = x.view(bs, 16, 1000, 10)\nprint(reshaped_tensor.shape)\nreshaped_and_permuted_tensor = reshaped_tensor.permute(0, 1, 3, 2)\nreshaped_and_permuted_tensor = reshaped_and_permuted_tensor.reshape(bs, 16 * 10, 1000)\n\n# チャネル次元の追加と複製\nx = torch.unsqueeze(reshaped_and_permuted_tensor, dim=1)\n\nx = torch.cat([x, x, x], dim=1)\n\nprint(f'Final shape: {x.shape}')\n# 最終的な形状は、(bs, 3, 160, 1000) となる","metadata":{"execution":{"iopub.status.busy":"2024-04-10T01:37:42.060575Z","iopub.execute_input":"2024-04-10T01:37:42.061044Z","iopub.status.idle":"2024-04-10T01:37:42.260930Z","shell.execute_reply.started":"2024-04-10T01:37:42.060995Z","shell.execute_reply":"2024-04-10T01:37:42.259726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# eeg 1d to 2d","metadata":{}},{"cell_type":"code","source":"def eeg_from_parquet(parquet_path):\n\n    eeg = pd.read_parquet(parquet_path, columns=FEATS2)\n    rows = len(eeg)\n    offset = (rows-10_000)//2\n    eeg = eeg.iloc[offset:offset+10_000]\n    data = np.zeros((10_000,len(FEATS2)))\n    for j,col in enumerate(FEATS2):\n        \n        # FILL NAN\n        x = eeg[col].values.astype('float32')\n        m = np.nanmean(x)\n        if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n        else: x[:] = 0\n        \n        data[:,j] = x\n\n    return data\n\ndef eeg_from_npy(eeg, eeg_id, OUTPUT_DIR, display=False):\n    \n    # EXTRACT MIDDLE 50 SECONDS\n#     eeg = eegs\n#     rows = len(eeg)\n#     offset = (rows-10_000)//2\n#     eeg = eeg.iloc[offset:offset+10_000]\n    colors = [\"#FF0000\", \"#00FF00\", \"#0000FF\"]\n    \n    if display: \n        plt.figure(figsize=(10,5), num=1, clear=True)\n        offset = 500\n    \n    x = eeg.copy()\n        \n    x = butter_lowpass_filter(x)\n    \n    \n    data = np.zeros((2000,len(FEATS)))\n    for j,col in enumerate(range(x.shape[1])):\n        \n        # FILL NAN\n#         x = eeg[col].values.astype('float32')\n#         m = np.nanmean(x)\n#         if np.isnan(x).mean()<1: x = np.nan_to_num(x,nan=m)\n#         else: x[:] = 0\n\n        _x = x[:,j]\n        \n        \n        if display: \n#             if j!=0: offset += x.max()\n            if j!=0: offset += 700\n#             offset += 300\n            plt.plot(range(2_000),(_x*100)+offset,label=col, color = colors[j%3])\n#             print(max(x), min(x))\n#             offset -= x.min()\n    data = x\n    if display:\n        # plt.legend()\n#         name = parquet_path.split('/')[-1]\n#         name = name.split('.')[0]\n        plt.gca().spines['right'].set_visible(False)\n        plt.gca().spines['top'].set_visible(False)\n        plt.gca().spines['bottom'].set_visible(False)\n        plt.gca().spines['left'].set_visible(False)\n        plt.xticks(color=\"None\")\n        plt.yticks(color=\"None\")\n        plt.tick_params(length=0)\n        plt.ylim(0,5000)\n        # plt.title(f'EEG {name}',size=16)\n        # plt.show()\n        strFile = f'{OUTPUT_DIR}/{eeg_id}.png'\n        if os.path.isfile(strFile):\n           os.remove(strFile) \n        plt.savefig(strFile)\n        plt.cla()\n        # plt.clf()\n        \n    return data\n\n\ndef generate_raw(eeg):\n\n    X = np.zeros((10_000,8),dtype='float32')\n    y = np.zeros((6,),dtype='float32')\n\n#     row = data.iloc[index]\n#     eeg = raw_eegs[row.eeg_id]\n\n    # FEATURE ENGINEER\n    X[:,0] = eeg[:,FEAT2IDX['Fp1']] - eeg[:,FEAT2IDX['T3']]\n    X[:,1] = eeg[:,FEAT2IDX['T3']] - eeg[:,FEAT2IDX['O1']]\n\n    X[:,2] = eeg[:,FEAT2IDX['Fp1']] - eeg[:,FEAT2IDX['C3']]\n    X[:,3] = eeg[:,FEAT2IDX['C3']] - eeg[:,FEAT2IDX['O1']]\n\n    X[:,4] = eeg[:,FEAT2IDX['Fp2']] - eeg[:,FEAT2IDX['C4']]\n    X[:,5] = eeg[:,FEAT2IDX['C4']] - eeg[:,FEAT2IDX['O2']]\n\n    X[:,6] = eeg[:,FEAT2IDX['Fp2']] - eeg[:,FEAT2IDX['T4']]\n    X[:,7] = eeg[:,FEAT2IDX['T4']] - eeg[:,FEAT2IDX['O2']]\n\n    # STANDARDIZE\n    X = np.clip(X,-1024,1024)\n    X = np.nan_to_num(X, nan=0) / 32.0\n\n    # BUTTER LOW-PASS FILTER\n    X = butter_lowpass_filter(X)\n    # Downsample\n    X = X[::5,:]\n\n    return X,y\n\ndef butter_lowpass_filter(data, cutoff_freq=20, sampling_rate=200, order=4):\n    nyquist = 0.5 * sampling_rate\n    normal_cutoff = cutoff_freq / nyquist\n    b, a = butter(order, normal_cutoff, btype='low', analog=False)\n    filtered_data = lfilter(b, a, data, axis=0)\n    return filtered_data\n\nuse_num_head = True\ntrain = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\nall_raw_eegs2 = {}\nif use_num_head:\n    EEG_IDS2 = [train.eeg_id.unique()[0]]\nFEATS2 = ['Fp1','T3','C3','O1','Fp2','C4','T4','O2']\nFEAT2IDX = {x:y for x,y in zip(FEATS2,range(len(FEATS2)))}\nFEATS = [['Fp1','F7','T3','T5','O1'],\n         ['Fp1','F3','C3','P3','O1'],\n         ['Fp2','F8','T4','T6','O2'],\n         ['Fp2','F4','C4','P4','O2']]\nPATH2 = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs'\n\nprint('Processing Test EEG parquets...'); print()\nfor i, eeg_id in enumerate(EEG_IDS2):\n    data = eeg_from_parquet(f'{PATH2}/{eeg_id}.parquet')\n    all_raw_eegs2[eeg_id] = data\n    \nfor i,(eeg_id) in tqdm(enumerate(EEG_IDS2), total=len(EEG_IDS2)):\n    raw_eeg = all_raw_eegs2[eeg_id]\n    X = generate_raw(raw_eeg)[0]\n    data = eeg_from_npy(X, eeg_id, \n                        \"/kaggle/working\", False)\n    \nprint(data.shape)","metadata":{"execution":{"iopub.status.busy":"2024-04-10T01:37:42.263860Z","iopub.execute_input":"2024-04-10T01:37:42.264251Z","iopub.status.idle":"2024-04-10T01:37:42.830169Z","shell.execute_reply.started":"2024-04-10T01:37:42.264218Z","shell.execute_reply":"2024-04-10T01:37:42.828753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#PythonでPyTorchを使用したサンプルコードを書いてみます。このコードでは、初めに仮想のEEGデータを作成し、その後に説明した形状変更とチャネルの追加処理を行います。最初のステップとして、バッチサイズbs、チャネル数16、時間軸10000のデータを生成し、それを画像データとして扱うための変換を実施します。\nimport torch\n\n# バッチサイズ、チャネル数、時間軸の長さを設定\nbs = 1  # 例えば、バッチサイズを64とする\nchannels = data.shape[1]\ntime_length = data.shape[0]  # 時間軸の長さ\n\n# 仮想のEEGデータを生成 (形状: bs x channels x time_length)\n#x = torch.randn(bs, channels, time_length)\nx = torch.tensor(data)\nx = x.permute(1,0)\nx = x.unsqueeze(0)\nprint(x.shape)\n\n# データを2D画像のように扱うために次元の拡張を行う\n# 実際には、ここでのコメントに対応するコードが省略されていますが、\n# x = x.unsqueeze(1)  # 形状を (bs, 1, 16, 10000) に変更\n\n# 形状変更と次元の入れ替え\nreshaped_tensor = x.view(bs, channels, int(time_length/10), 10)\nprint(reshaped_tensor.shape)\nreshaped_and_permuted_tensor = reshaped_tensor.permute(0, 1, 3, 2)\nprint(reshaped_and_permuted_tensor.shape)\nreshaped_and_permuted_tensor = reshaped_and_permuted_tensor.reshape(bs, channels * 10, int(time_length/10))\nprint(reshaped_and_permuted_tensor.shape)\n\n# チャネル次元の追加と複製\nx = torch.unsqueeze(reshaped_and_permuted_tensor, dim=1)\n\nx = torch.cat([x, x, x], dim=1)\n\nprint(f'Final shape: {x.shape}')\n# 最終的な形状は、(bs, 3, 160, 1000) となる","metadata":{"execution":{"iopub.status.busy":"2024-04-10T01:37:42.831971Z","iopub.execute_input":"2024-04-10T01:37:42.832489Z","iopub.status.idle":"2024-04-10T01:37:42.865251Z","shell.execute_reply.started":"2024-04-10T01:37:42.832427Z","shell.execute_reply":"2024-04-10T01:37:42.863819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#波形と画像の比較\n\n#波形\ndf = pd.DataFrame(data)\nfig, axes = plt.subplots(nrows=2, ncols=4, figsize=(20, 10))\nfig.tight_layout(pad=3.0)\nfor i, c in enumerate(df.columns):\n    row = i // 4\n    col = i % 4\n    axes[row, col].set_title(c)\n    df[c].plot(ax=axes[row, col])\n\nplt.show()\n\n#画像\nx_np = x.squeeze(0).numpy()  # バッチ次元を削除\n\n# imshowが期待する形状に配列を転置\n# 現在の形状が(3, 80, 200)なので、(80, 200, 3)に変更\nx_np_transposed = np.transpose(x_np, (1, 2, 0))\n# 修正された配列で画像を表示\nplt.imshow(x_np_transposed)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-10T01:42:09.567178Z","iopub.execute_input":"2024-04-10T01:42:09.567604Z","iopub.status.idle":"2024-04-10T01:42:11.725251Z","shell.execute_reply.started":"2024-04-10T01:42:09.567571Z","shell.execute_reply":"2024-04-10T01:42:11.723977Z"},"trusted":true},"execution_count":null,"outputs":[]}]}