{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\nimport os\nimport pyarrow.parquet as pq\nimport glob\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import preprocessing\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-13T08:17:46.378758Z","iopub.execute_input":"2024-03-13T08:17:46.379098Z","iopub.status.idle":"2024-03-13T08:17:50.129152Z","shell.execute_reply.started":"2024-03-13T08:17:46.379069Z","shell.execute_reply":"2024-03-13T08:17:50.128089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(plt.style.available)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T06:18:33.172236Z","iopub.execute_input":"2024-03-13T06:18:33.172758Z","iopub.status.idle":"2024-03-13T06:18:33.179733Z","shell.execute_reply.started":"2024-03-13T06:18:33.172726Z","shell.execute_reply":"2024-03-13T06:18:33.178419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('seaborn-v0_8-deep')","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:17:56.811429Z","iopub.execute_input":"2024-03-13T08:17:56.811906Z","iopub.status.idle":"2024-03-13T08:17:56.817853Z","shell.execute_reply.started":"2024-03-13T08:17:56.811883Z","shell.execute_reply":"2024-03-13T08:17:56.816245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_eeg_files = glob.glob('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/*')\ntest_eeg_files = glob.glob('/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/*')\ntrain_spectrogram_files = glob.glob('/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/*')\ntest_spectrogram_files = glob.glob('/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/*')\nprint(f'Total number of files in train_eegs: {len(train_eeg_files)}')\nprint(f'Total number of files in test_eegs: {len(test_eeg_files)}')\nprint(f'Total number of files in train_spectrograms: {len(train_spectrogram_files)}')\nprint(f'Total number of files in test_spectrograms: {len(test_spectrogram_files)}')","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:18:03.315290Z","iopub.execute_input":"2024-03-13T08:18:03.315617Z","iopub.status.idle":"2024-03-13T08:18:03.753713Z","shell.execute_reply.started":"2024-03-13T08:18:03.315593Z","shell.execute_reply":"2024-03-13T08:18:03.752564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ntest_csv = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nsample_csv = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv')\nprint(f'number of entries in train_csv {len(train_csv)}')\nprint(f'number of entries in test_csv {len(test_csv)}')\nprint(f'number of entries in sample_csv {len(sample_csv)}')","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:18:04.921832Z","iopub.execute_input":"2024-03-13T08:18:04.922170Z","iopub.status.idle":"2024-03-13T08:18:05.111589Z","shell.execute_reply.started":"2024-03-13T08:18:04.922144Z","shell.execute_reply":"2024-03-13T08:18:05.110523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train = train_csv.drop(['eeg_label_offset_seconds',\n                'spectrogram_id',\n                'spectrogram_sub_id',\n                'spectrogram_label_offset_seconds',\n                'label_id',\n                'patient_id',\n                'expert_consensus'], axis=1)\n\nnew_train = new_train[new_train['eeg_sub_id']==0].drop('eeg_sub_id', axis=1)\n\nnew_train.reset_index(drop=True, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:18:10.544602Z","iopub.execute_input":"2024-03-13T08:18:10.544947Z","iopub.status.idle":"2024-03-13T08:18:10.680923Z","shell.execute_reply.started":"2024-03-13T08:18:10.544920Z","shell.execute_reply":"2024-03-13T08:18:10.679928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_train","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:18:11.911987Z","iopub.execute_input":"2024-03-13T08:18:11.912613Z","iopub.status.idle":"2024-03-13T08:18:11.930396Z","shell.execute_reply.started":"2024-03-13T08:18:11.912584Z","shell.execute_reply":"2024-03-13T08:18:11.929368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_train = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\ninput_data_file_names = new_train['eeg_id']\ny_data = new_train.iloc[:, 1:].to_numpy()","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:18:17.773471Z","iopub.execute_input":"2024-03-13T08:18:17.773821Z","iopub.status.idle":"2024-03-13T08:18:17.779712Z","shell.execute_reply.started":"2024-03-13T08:18:17.773794Z","shell.execute_reply":"2024-03-13T08:18:17.778773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_divisible_by = []\nfor n in range(100):\n    if (len(input_data_file_names) % (n+1)) == 0:\n        data_divisible_by.append(n+1)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:18:20.413650Z","iopub.execute_input":"2024-03-13T08:18:20.414892Z","iopub.status.idle":"2024-03-13T08:18:20.420413Z","shell.execute_reply.started":"2024-03-13T08:18:20.414844Z","shell.execute_reply":"2024-03-13T08:18:20.418850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_divisible_by","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:18:21.192847Z","iopub.execute_input":"2024-03-13T08:18:21.193172Z","iopub.status.idle":"2024-03-13T08:18:21.199912Z","shell.execute_reply.started":"2024-03-13T08:18:21.193144Z","shell.execute_reply":"2024-03-13T08:18:21.198661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(input_data_file_names) / 23","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:18:23.560573Z","iopub.execute_input":"2024-03-13T08:18:23.560954Z","iopub.status.idle":"2024-03-13T08:18:23.567350Z","shell.execute_reply.started":"2024-03-13T08:18:23.560926Z","shell.execute_reply":"2024-03-13T08:18:23.566224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch = int(len(input_data_file_names) / 23)\nprint(batch)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:18:25.612855Z","iopub.execute_input":"2024-03-13T08:18:25.613174Z","iopub.status.idle":"2024-03-13T08:18:25.618330Z","shell.execute_reply.started":"2024-03-13T08:18:25.613147Z","shell.execute_reply":"2024-03-13T08:18:25.617153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_batch = 1\nx_data_normalized = []\nfor idx in range((n_batch-1)*batch, n_batch*batch):\n    #print(idx)\n    path = os.path.join(root_train, str(input_data_file_names[idx]) + '.parquet')\n    table = pq.read_table(path).to_pandas().fillna(0).to_numpy()\n    table_normalized = preprocessing.normalize(table)\n    x_data_normalized.append(table_normalized)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:18:45.035037Z","iopub.execute_input":"2024-03-13T08:18:45.035395Z","iopub.status.idle":"2024-03-13T08:19:03.977522Z","shell.execute_reply.started":"2024-03-13T08:18:45.035369Z","shell.execute_reply":"2024-03-13T08:19:03.976463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_data_normalized = preprocessing.normalize(y_data[(n_batch-1)*batch: n_batch*batch], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:19:05.525655Z","iopub.execute_input":"2024-03-13T08:19:05.525989Z","iopub.status.idle":"2024-03-13T08:19:05.531480Z","shell.execute_reply.started":"2024-03-13T08:19:05.525963Z","shell.execute_reply":"2024-03-13T08:19:05.530511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(x_data_normalized, y_data_normalized, test_size=0.1, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:19:30.632072Z","iopub.execute_input":"2024-03-13T08:19:30.632417Z","iopub.status.idle":"2024-03-13T08:19:30.638994Z","shell.execute_reply.started":"2024-03-13T08:19:30.632391Z","shell.execute_reply":"2024-03-13T08:19:30.637971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train[0][:, 0].shape","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:19:55.187806Z","iopub.execute_input":"2024-03-13T08:19:55.188164Z","iopub.status.idle":"2024-03-13T08:19:55.195576Z","shell.execute_reply.started":"2024-03-13T08:19:55.188136Z","shell.execute_reply":"2024-03-13T08:19:55.194564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"von_sampel = int(len(X_train[0][:, 0])/6)\ns_sampled_data = []\nfor sampel in range(6):\n    bruh = X_train[0][sampel*von_sampel:(sampel+1)*von_sampel, :]\n    s_sampled_data.append(bruh)","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:55:09.924916Z","iopub.execute_input":"2024-03-13T08:55:09.925316Z","iopub.status.idle":"2024-03-13T08:55:09.936871Z","shell.execute_reply.started":"2024-03-13T08:55:09.925284Z","shell.execute_reply":"2024-03-13T08:55:09.935502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for sampled in s_sampled_data:\n    print(sampled.mean(0))","metadata":{"execution":{"iopub.status.busy":"2024-03-13T09:34:00.178231Z","iopub.execute_input":"2024-03-13T09:34:00.178583Z","iopub.status.idle":"2024-03-13T09:34:00.186152Z","shell.execute_reply.started":"2024-03-13T09:34:00.178557Z","shell.execute_reply":"2024-03-13T09:34:00.185164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train[0].shape","metadata":{"execution":{"iopub.status.busy":"2024-03-13T08:20:37.535412Z","iopub.execute_input":"2024-03-13T08:20:37.535780Z","iopub.status.idle":"2024-03-13T08:20:37.542022Z","shell.execute_reply.started":"2024-03-13T08:20:37.535754Z","shell.execute_reply":"2024-03-13T08:20:37.540942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}