{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport pyarrow.parquet as pq\nimport numpy as np\nimport pickle\nfrom sklearn.preprocessing import MinMaxScaler\nimport os\nfrom sklearn.impute import SimpleImputer\nimport torch\nfrom torch import cuda","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-02T05:36:34.798810Z","iopub.execute_input":"2024-04-02T05:36:34.799175Z","iopub.status.idle":"2024-04-02T05:37:01.389865Z","shell.execute_reply.started":"2024-04-02T05:36:34.799144Z","shell.execute_reply":"2024-04-02T05:37:01.388759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip uninstall cupy -y #kaggle엔 cuda 12.1이 깔려 있어서 cupy를 버전에 맞게 다시 깔아줘야 함. 패키지 충돌 방지를 위해 기존 cupy 삭제\n!pip install cupy-cuda12x\n!pip install torch-geometric","metadata":{"execution":{"iopub.status.busy":"2024-04-02T05:37:01.392265Z","iopub.execute_input":"2024-04-02T05:37:01.392916Z","iopub.status.idle":"2024-04-02T05:37:19.558116Z","shell.execute_reply.started":"2024-04-02T05:37:01.392866Z","shell.execute_reply":"2024-04-02T05:37:19.557087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch_geometric.data import Data\nfrom torch_geometric.loader import DataLoader\nfrom tqdm import tqdm\nimport cupy as cp","metadata":{"execution":{"iopub.status.busy":"2024-04-02T05:47:40.815829Z","iopub.execute_input":"2024-04-02T05:47:40.817113Z","iopub.status.idle":"2024-04-02T05:47:40.821312Z","shell.execute_reply.started":"2024-04-02T05:47:40.817050Z","shell.execute_reply":"2024-04-02T05:47:40.820613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_correlation(data_dir, p, start = None, during=None ):\n    eeg = pq.read_table(data_dir) \n    df_eeg = eeg.to_pandas() \n    spectro_data = df_eeg[ df_eeg.columns.difference(['time'])] #time 열 삭제\n    \n    channel_list = spectro_data.columns #channel명들 리스트 반환\n    num_channels = len(channel_list)\n    \n    if (start is None) and (during is None):\n        count = 1 #10분 spectrogram을 segmentation할 거임. 나눠지는 segment 개수 세아리기 시작\n        print(\"start making correlation\")\n        corr_array = cp.zeros((400,400), dtype=cp.float32) #correlation matrix만들 빈 행렬 생성\n        avg_eeg_c = cp.empty([400, 68])\n        for t_length in range(0, len(spectro_data), 68): #전체 time point에 대하여 segmentation 시작      \n            print(\"section \", count)\n            eeg_data = spectro_data.iloc[t_length:t_length+68]\n            if len(eeg_data) < 34:\n                print(\"passed\")\n                pass\n            else: \n                X = np.array(eeg_data.transpose())\n                if 68-len(eeg_data) != 0:\n                    mean = X.mean(axis = 1).reshape(-1, 1)\n                    for apd in range(68-len(eeg_data)):\n                        X = np.concatenate((X, mean), axis=1)\n                eeg_cp = np.asarray(X) #array gpu로 올려서 계산\n                corr = np.corrcoef(eeg_cp)\n                corr[np.where(corr < p)] = 0 # 상관관계 계수가 threshold보다 작으면 0\n                avg_eeg_c += eeg_cp \n                corr_array += corr\n                count += 1\n        avg_eeg_c /= count #eeg data 평균\n        corr_array /= count #correlation matrix 평균\n        print(\"complete making correlation\")\n\n        return avg_eeg_c, corr_array\n    else:\n        eegs_cp, corrs = [], []\n        for st, dur in zip(start_list, during_list):  #start와 during은 subject 내에 class가 여러개 있을 때만 입력됨. 즉, subject class가 여러 개 일때\n            if dur < 68 : #지속 시간이 기존 segment의 절반 미만이면 pass\n                pass\n            else:\n                corr_array = np.zeros((400,400))\n                eeg_data = spectro_data.iloc[int(st/2) : int((st+dur)/2)]\n                avg_eeg_c = np.empty([400,68])\n                count = 1\n                for t_length in range(0, len(spectro_data), 68):\n                    eeg_data = eeg_data.iloc[t_length:t_length+68]\n                    if len(eeg_data) < 46:\n                        print(\"passed\")\n                        pass\n\n                    else: \n                        print(\"section \", count)\n                        X = np.array(eeg_data.transpose())\n                        if (68-len(eeg_data)) != 0:\n                            mean = X.mean(axis = 1).reshape(-1, 1)\n                            for apd in range(68-len(eeg_data)):\n                                X = np.concatenate((X, mean), axis=1)\n                        corr = np.corrcoef(X)\n                        corr[np.where(corr < p)] = 0 # 상관관계 계수가 threshold보다 작으면 0\n                        avg_eeg_c += X\n                        corr_array += corr\n                        count += 1\n\n                eegs_cp.append(avg_eeg_c / count)\n                corrs.append(corr_array / count)\n\n        print(\"complete making correlation\")\n    return eegs_cp, corrs","metadata":{"execution":{"iopub.status.busy":"2024-04-02T05:57:05.375584Z","iopub.execute_input":"2024-04-02T05:57:05.376007Z","iopub.status.idle":"2024-04-02T05:57:05.389484Z","shell.execute_reply.started":"2024-04-02T05:57:05.375972Z","shell.execute_reply":"2024-04-02T05:57:05.388760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_graph(data, corr_matrix, Y):\n    if len(data)==0 :\n        return 0\n    else:\n        print(\"start making Graph\")\n        nodes = []\n        edges = [[], []]\n        attr = []\n        state = 0\n        for i in range(corr_matrix.shape[0]):\n            if np.isnan(data[i]).all(): #.get()은 gpu로 올렸던 array를 다시 cpu로 불러옴. 모든 값이 nan이면 \n                state = 1 #state 1\n                break\n\n            elif np.isnan((data[i]).reshape(-1,1)).any():\n                nan_mask = np.isnan((data).reshape(-1,1)).any(axis=1) \n                nan_row_count = np.count_nonzero(nan_mask)\n                if len((data).reshape(-1,1))/3 < nan_row_count : #nan값이 3분의 1 이상이면\n                    state = 1 # state 1\n                    break\n                else:\n                    imputer = SimpleImputer(strategy='mean') #아니면 mean값으로 채움\n                    X = imputer.fit_transform((data[i]).reshape(-1,1))\n            else :\n                X = (data[i]).reshape(-1,1) #값이 온전히 다 있을 때\n            scaler = MinMaxScaler()\n            minmax_signal = scaler.fit_transform(X) #minmax\n            nodes.append(minmax_signal) #노드 특성에 시계열값 입력\n            for j in range(corr_matrix.shape[1]):\n                if i > j :\n                    if  corr_matrix[i, j] != 0:\n                        edges[0].append(i) #엣지에 연결된 노드들 추가\n                        edges[0].append(j)\n                        edges[1].append(j) #엣지 특성(correlation 값 입력)\n                        edges[1].append(i)\n\n                        attr.append(corr_matrix[i, j]) #gpu -> cpu 나중에 연산할때 문제됨\n                        attr.append(corr_matrix[j, i])\n        if state == 1:\n            return 0\n\n        else :\n\n            nodes = torch.tensor(np.array(nodes)) #tensor로 바꿔서\n            edges = torch.tensor(np.array(edges))\n            edge_attr = torch.tensor(np.array(attr))\n\n            class_list = ['GPD', 'GRDA', 'LPD', 'LRDA', 'Other', 'Seizure']\n\n            y = class_list.index(Y)/len(class_list)\n\n            G = Data(x = nodes, edge_index = edges, edge_attr = attr, y = y) #저장\n\n            print(\"complete making graph\")\n\n            return G","metadata":{"execution":{"iopub.status.busy":"2024-04-02T08:41:02.931527Z","iopub.execute_input":"2024-04-02T08:41:02.932280Z","iopub.status.idle":"2024-04-02T08:41:02.983591Z","shell.execute_reply.started":"2024-04-02T08:41:02.932239Z","shell.execute_reply":"2024-04-02T08:41:02.982204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_list = os.listdir(\"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms\")\nmeta_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\n\na = 0\nb = 0\n\ncheck = {}\n\nfor eeg in eeg_list :\n    ID = int(eeg.split('.')[0])\n    Y = meta_df[\"expert_consensus\"].iloc[meta_df.index[(meta_df[\"spectrogram_id\"] == ID)]]\n    if all(x == Y.iloc[0] for x in Y.iloc[1:]):\n        a += 1\n    else:\n        b+= 1\n        check[ID] = Y.unique()\n    \n\n\nprint(\"same : \", a, \"\\ndifferent :\", b)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T05:53:15.171372Z","iopub.execute_input":"2024-04-02T05:53:15.172315Z","iopub.status.idle":"2024-04-02T05:53:20.068097Z","shell.execute_reply.started":"2024-04-02T05:53:15.172277Z","shell.execute_reply":"2024-04-02T05:53:20.067365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ngraph_list = []\n\n_dir= \"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms\"\nfile_list = os.listdir(_dir)\nmeta_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\n\nfor filename in tqdm(file_list):\n    file_dir = _dir + '/' + filename\n    if int(filename.split(\".\")[0]) in check.keys():\n        print(\"start\", filename)\n        subdf = meta_df[meta_df['spectrogram_id'] == int(filename.split('.')[0])]\n        cons_list = subdf['expert_consensus'].unique()\n        for cons in set(cons_list):\n            print(cons)\n            \n            a = subdf[subdf['expert_consensus'] == cons]\n            b = a[\"spectrogram_sub_id\"]\n            id_check_start = [b.iloc[0]]\n            id_check_finish = [b.iloc[-1]]\n            state = 0\n            for sub_id in range(len(b)-1):\n                if b.iloc[sub_id+1] - b.iloc[sub_id] != 1:\n                    state = 1\n                    id_check_finish.append(b.iloc[sub_id])\n                    id_check_start.append(b.iloc[sub_id+1])\n            id_check_start.sort()\n            id_check_finish.sort()\n            assert len(id_check_start) is len(id_check_finish)\n            \n            start_list = []\n            during_list = []\n            for st, fn in zip(id_check_start, id_check_finish):\n                offset = subdf['spectrogram_label_offset_seconds']\n                start_time = offset[subdf[\"spectrogram_sub_id\"] == st].values\n                during = offset[subdf[\"spectrogram_sub_id\"] == fn].values - offset[subdf[\"spectrogram_sub_id\"] == st].values\n                start_list.append(start_time.astype(int)[0])\n                during_list.append(during.astype(int)[0])\n            \n            spectro, corr = make_correlation(data_dir = file_dir, start = start_list, during = during_list , p = 0.14)\n            \n            for sp, cr, i in zip(spectro, corr, range(len(spectro))):\n                G = make_graph(sp, cr, cons)\n                if G == 0:\n                    print(\"%s_%s_%d passed\"%(filename.split(\".\")[0],cons, i+1))\n                    break\n                else:\n                    graph_list.append(G)\n                    with open(\"/kaggle/working/graph_%s_%s_%d\"%(filename.split(\".\")[0],cons, i+1) , \"wb\") as fw:\n                        pickle.dump(G, fw)\n\n        print(\"%s finished\"%filename)\n    \nwith open(\"/kaggle/working/graph_list\" , \"wb\") as p:\n    pickle.dump(graph_list, p)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T05:57:18.884827Z","iopub.execute_input":"2024-04-02T05:57:18.885660Z","iopub.status.idle":"2024-04-02T05:57:37.651266Z","shell.execute_reply.started":"2024-04-02T05:57:18.885619Z","shell.execute_reply":"2024-04-02T05:57:37.650387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" '''\n    if (int(filename.split('.')[0]) not in list(check.keys())):\n    \n        y_key = (meta_df[\"spectrogram_id\"] == int(filename.split(\".\")[0]))\n        Y = meta_df[\"expert_consensus\"].iloc[meta_df.index[y_key]]\n\n        cons = Y.iloc[0]\n        assert type(cons) == str\n\n        spectro, corr = make_correlation(data_dir = file_dir, start = None, during=None , p = 0.14)\n\n        G = make_graph(spectro, corr, cons)\n\n        with open(\"/kaggle/working/graph_%s\"%filename.split(\".\")[0] , \"wb\") as fw:\n            pickle.dump(G, fw)\n\n        print(\"%s finished\"%filename)\n        graph_list.append(G)\n    '''","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filename= \"1780849718.parquet\"","metadata":{"execution":{"iopub.status.busy":"2024-04-02T05:56:19.159079Z","iopub.execute_input":"2024-04-02T05:56:19.159549Z","iopub.status.idle":"2024-04-02T05:56:19.163777Z","shell.execute_reply.started":"2024-04-02T05:56:19.159511Z","shell.execute_reply":"2024-04-02T05:56:19.162981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\n_dir= \"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms\"\nmeta_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\nfile_dir = _dir + '/' + filename\nprint(\"start\", filename)\nsubdf = meta_df[meta_df['spectrogram_id'] == int(filename.split('.')[0])]\ncons_list = subdf['expert_consensus'].unique()\nfor cons in set(cons_list):\n    print(cons)\n\n    a = subdf[subdf['expert_consensus'] == cons]\n    b = a[\"spectrogram_sub_id\"]\n    id_check_start = [b.iloc[0]]\n    id_check_finish = [b.iloc[-1]]\n    state = 0\n    for sub_id in range(len(b)-1):\n        if b.iloc[sub_id+1] - b.iloc[sub_id] != 1:\n            state = 1\n            id_check_finish.append(b.iloc[sub_id])\n            id_check_start.append(b.iloc[sub_id+1])\n    id_check_start.sort()\n    id_check_finish.sort()\n    assert len(id_check_start) is len(id_check_finish)\n\n    start_list = []\n    during_list = []\n    for st, fn in zip(id_check_start, id_check_finish):\n        offset = subdf['spectrogram_label_offset_seconds']\n        start_time = offset[subdf[\"spectrogram_sub_id\"] == st].values\n        during = offset[subdf[\"spectrogram_sub_id\"] == fn].values - offset[subdf[\"spectrogram_sub_id\"] == st].values\n        start_list.append(start_time.astype(int)[0])\n        during_list.append(during.astype(int)[0])\n        '''","metadata":{"execution":{"iopub.status.busy":"2024-04-02T05:56:27.244310Z","iopub.execute_input":"2024-04-02T05:56:27.244699Z","iopub.status.idle":"2024-04-02T05:56:27.384798Z","shell.execute_reply.started":"2024-04-02T05:56:27.244668Z","shell.execute_reply":"2024-04-02T05:56:27.383984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\neeg = pq.read_table(file_dir) \ndf_eeg = eeg.to_pandas() \nspectro_data = df_eeg[ df_eeg.columns.difference(['time'])] #time 열 삭제\np = 0.14\nchannel_list = spectro_data.columns #channel명들 리스트 반환\nnum_channels = len(channel_list)\neegs_cp, corrs = [],[]\nfor st, dur in zip(start_list, during_list):  #start와 during은 subject 내에 class가 여러개 있을 때만 입력됨. 즉, subject class가 여러 개 일때\n    if dur < 68 : #지속 시간이 기존 segment의 절반 미만이면 pass\n        pass\n    else:\n        corr_array = np.zeros((400,400))\n        eeg_data = spectro_data.iloc[int(st/2) : int((st+dur)/2)]\n        avg_eeg_c = np.empty([400,68])\n        count = 1\n        for t_length in range(0, len(spectro_data), 68):\n            eeg_data = eeg_data.iloc[t_length:t_length+68]\n            if len(eeg_data) < 46:\n                print(\"passed\")\n                pass\n\n            else: \n                print(\"section \", count)\n                X = np.array(eeg_data.transpose())\n                if 68-len(eeg_data) != 0:\n                    mean = X.mean(axis = 1).reshape(-1, 1)\n                    for apd in range(68-len(eeg_data)):\n                        X = np.concatenate((X, mean), axis=1)\n                corr = np.corrcoef(X)\n                corr[np.where(corr < p)] = 0 # 상관관계 계수가 threshold보다 작으면 0\n                avg_eeg_c += X\n                corr_array += corr\n                count += 1\n\n        eegs_cp.append(avg_eeg_c / count)\n        corrs.append(corr_array / count)\n        \n'''","metadata":{"execution":{"iopub.status.busy":"2024-04-02T05:56:40.756044Z","iopub.execute_input":"2024-04-02T05:56:40.756798Z","iopub.status.idle":"2024-04-02T05:56:40.809792Z","shell.execute_reply.started":"2024-04-02T05:56:40.756761Z","shell.execute_reply":"2024-04-02T05:56:40.809087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}