{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30626,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport torch\nfrom torch.utils.data import Dataset\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n#    for filename in filenames:\n#        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-10T23:39:35.576907Z","iopub.execute_input":"2024-01-10T23:39:35.577460Z","iopub.status.idle":"2024-01-10T23:39:39.813587Z","shell.execute_reply.started":"2024-01-10T23:39:35.577416Z","shell.execute_reply":"2024-01-10T23:39:39.812349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## load data","metadata":{}},{"cell_type":"code","source":"train_df =  pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nfor i, k in enumerate(train_df.columns):\n    print(i,k)","metadata":{"execution":{"iopub.status.busy":"2024-01-10T23:39:39.815608Z","iopub.execute_input":"2024-01-10T23:39:39.816128Z","iopub.status.idle":"2024-01-10T23:39:40.145462Z","shell.execute_reply.started":"2024-01-10T23:39:39.816077Z","shell.execute_reply":"2024-01-10T23:39:40.144209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.iloc[10,:]","metadata":{"execution":{"iopub.status.busy":"2024-01-10T23:39:40.146987Z","iopub.execute_input":"2024-01-10T23:39:40.147359Z","iopub.status.idle":"2024-01-10T23:39:40.165970Z","shell.execute_reply.started":"2024-01-10T23:39:40.147323Z","shell.execute_reply":"2024-01-10T23:39:40.164387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## create a dataloader ","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nfrom torchvision.io import read_image\n\nclass HMSDatasetTrain(Dataset):\n    def __init__(self, annotations_file, eeg_dir, train=True, transform=None, target_transform=None):\n        self.df = pd.read_csv(annotations_file)\n        self.eeg_dir = eeg_dir\n        self.train = True\n        self.transform = transform\n        self.target_transform = target_transform\n        #self.eeg_idx = self.df.columns.get_loc('eeg_id') # we do not have the same column index in train and in test\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        \n        eeg_path = os.path.join(self.eeg_dir, str(self.df.iloc[idx, 0]))+'.parquet'\n        # read the parquet file \n        eeg = pd.read_parquet(eeg_path)\n        eeg = eeg.values\n        offset = self.df.iloc[idx,2] # eeg_label_offset_seconds column\n        start = int(offset)*200# the sample rate is 200\n        duration = 50*200 # the duration is \n        eeg = eeg[start:start+duration,:]\n        label = self.df.iloc[idx, 9:]\n        label /=sum(label)\n        label = label.to_dict()\n        if self.transform:\n            eeg = self.transform(eeg)\n\n        if self.target_transform:\n            label = self.target_transform(label)\n        return eeg, label\n        \n\nannotations_file = '/kaggle/input/hms-harmful-brain-activity-classification/train.csv'\neeg_dir = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs'\ntraining_data = HMSDatasetTrain(annotations_file,eeg_dir )\neeg, label = training_data[11]\neeg.shape, label","metadata":{"execution":{"iopub.status.busy":"2024-01-10T23:39:40.169783Z","iopub.execute_input":"2024-01-10T23:39:40.170360Z","iopub.status.idle":"2024-01-10T23:39:41.111443Z","shell.execute_reply.started":"2024-01-10T23:39:40.170311Z","shell.execute_reply":"2024-01-10T23:39:41.110186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import DataLoader\n\ntrain_dataloader = DataLoader(training_data, batch_size=2, shuffle=False)\nnext(iter(train_dataloader))","metadata":{"execution":{"iopub.status.busy":"2024-01-10T23:39:41.112942Z","iopub.execute_input":"2024-01-10T23:39:41.113345Z","iopub.status.idle":"2024-01-10T23:39:41.296120Z","shell.execute_reply.started":"2024-01-10T23:39:41.113310Z","shell.execute_reply":"2024-01-10T23:39:41.294791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-01-10T23:39:41.298537Z","iopub.execute_input":"2024-01-10T23:39:41.298988Z","iopub.status.idle":"2024-01-10T23:39:41.323911Z","shell.execute_reply.started":"2024-01-10T23:39:41.298949Z","shell.execute_reply":"2024-01-10T23:39:41.322481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}