{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7465251,"sourceType":"datasetVersion","datasetId":4317718}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Possible augmentations for EEG Graph","metadata":{"execution":{"iopub.status.busy":"2024-02-28T01:22:00.065565Z","iopub.execute_input":"2024-02-28T01:22:00.066878Z","iopub.status.idle":"2024-02-28T01:22:01.848000Z","shell.execute_reply.started":"2024-02-28T01:22:00.066824Z","shell.execute_reply":"2024-02-28T01:22:01.846390Z"}}},{"cell_type":"markdown","source":"### - Goal of this notebooks is to try different augmentations with EEG data.\n### - I have already added 5 different possible augmentation techniques for EEG data. Where we just need one parquet file from which we are showing original EEG graph and augmented EEG graph.\n### - Try changing Augmentation functions with parameter's of the function to see augmented EEG graph.","metadata":{}},{"cell_type":"markdown","source":"## Step 1: Imports","metadata":{"execution":{"iopub.status.busy":"2024-03-04T11:59:47.925094Z","iopub.execute_input":"2024-03-04T11:59:47.925663Z","iopub.status.idle":"2024-03-04T11:59:47.955948Z","shell.execute_reply.started":"2024-03-04T11:59:47.925626Z","shell.execute_reply":"2024-03-04T11:59:47.954402Z"}}},{"cell_type":"code","source":"import torch\nimport pandas as pd\nimport numpy as np\nfrom matplotlib import pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2024-03-04T12:46:17.112292Z","iopub.execute_input":"2024-03-04T12:46:17.112723Z","iopub.status.idle":"2024-03-04T12:46:17.119191Z","shell.execute_reply.started":"2024-03-04T12:46:17.112689Z","shell.execute_reply":"2024-03-04T12:46:17.117157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 2: Features","metadata":{}},{"cell_type":"code","source":"EEG_PARQUET = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1000913311.parquet\"\nFEATS       = ['Fp1','T3','C3','O1','Fp2','C4','T4','O2']","metadata":{"execution":{"iopub.status.busy":"2024-03-04T12:46:26.454489Z","iopub.execute_input":"2024-03-04T12:46:26.454955Z","iopub.status.idle":"2024-03-04T12:46:26.460193Z","shell.execute_reply.started":"2024-03-04T12:46:26.454894Z","shell.execute_reply":"2024-03-04T12:46:26.459188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 3: Utilities","metadata":{}},{"cell_type":"code","source":"class UTITLITIES:\n    \n    def eeg_from_parquet(self, parquet_path, display=False):\n        # EXTRACT MIDDLE 50 SECONDS\n        eeg    = pd.read_parquet(parquet_path, columns=FEATS)\n        rows   = len(eeg)\n        offset = (rows-10_000)//2\n        eeg    = eeg.iloc[offset:offset+10_000]\n\n        if display: \n            plt.figure(figsize=(10,5))\n            offset = 0\n\n        # CONVERT TO NUMPY\n        data = np.zeros((10_000,len(FEATS)))\n\n        for j,col in enumerate(FEATS):\n\n            # FILL NAN\n            x = eeg[col].values.astype('float32')\n            m = np.nanmean(x)\n\n            if np.isnan(x).mean()<1: \n                x = np.nan_to_num(x,nan=m)\n            else: \n                x[:] = 0\n\n            data[:,j] = x\n\n            if display: \n                if j!=0: offset += x.max()\n                plt.plot(range(10_000),x-offset,label=col)\n                offset -= x.min()\n\n        if display:\n            plt.legend()\n            name = parquet_path.split('/')[-1]\n            name = name.split('.')[0]\n            plt.title(f'EEG {name}',size=16)\n            plt.show()\n\n        return data\n        \n        \n    def show_eeg_two(self, img1, img2, eeg_id1=None, eeg_id2=None):\n        fig, axes = plt.subplots(1, 2, figsize=(18, 6))  # Create subplots with 2 columns\n\n        electrode_names = ['Fp1', 'T3', 'C3', 'O1', 'Fp2', 'C4', 'T4', 'O2']\n\n        for ax, img, eeg_id in zip(axes, [img1, img2], [eeg_id1, eeg_id2]):\n            if isinstance(img, torch.Tensor):\n                img = img.detach().numpy()\n                img = img.transpose(1, 0)\n\n            offset = 0\n            for j in range(img.shape[-1]):\n                if j != 0: \n                    offset -= img[:, j].min()\n                ax.plot(img[:, j] + offset, label=f'feature {j+1}')\n                offset += img[:, j].max() + 1  # Adding 1 for visual separation\n\n            ax.legend(labels=['{} - {}'.format(label, electrode_names[i]) for i, label in enumerate(range(1, img.shape[-1] + 1))])\n            if eeg_id:\n                ax.set_title(f'EEG_Id = {eeg_id}', size=14)\n            else:\n                ax.set_title('EEG Data', size=14)\n\n            ax.set_xlabel('Time (samples)')\n            ax.set_ylabel('Amplitude')\n\n        plt.tight_layout()\n        plt.show()\n    \nutilities = UTITLITIES()","metadata":{"execution":{"iopub.status.busy":"2024-03-04T12:46:27.207335Z","iopub.execute_input":"2024-03-04T12:46:27.207731Z","iopub.status.idle":"2024-03-04T12:46:27.224645Z","shell.execute_reply.started":"2024-03-04T12:46:27.207700Z","shell.execute_reply":"2024-03-04T12:46:27.223074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 4: Augmentation class","metadata":{}},{"cell_type":"code","source":"class AUG:\n    def add_noise_to_eeg(aelf, eeg_data, noise_level=0.1):\n        \"\"\"\n        Add Gaussian noise to EEG data.\n\n        Args:\n        - eeg_data (np.ndarray): Input EEG data array with shape (10000, 8)\n        - noise_level (float): Level of Gaussian noise to add (default is 0.1)\n\n        Returns:\n        - noisy_eeg_data (np.ndarray): Noisy EEG data array with shape (10000, 8)\n        \"\"\"\n        # Convert numpy array to PyTorch tensor\n        eeg_tensor = torch.tensor(eeg_data, dtype=torch.float32)\n\n        # Generate noise with the same shape as the input EEG data\n        noise = torch.randn_like(eeg_tensor) * noise_level\n\n        # Add noise to the EEG data\n        noisy_eeg_tensor = eeg_tensor + noise\n\n        # Convert noisy EEG tensor back to numpy array\n        noisy_eeg_data = noisy_eeg_tensor.numpy()\n\n        return noisy_eeg_data\n\n    def introduce_time_shifts(self, eeg_data, max_shift=10):\n        \"\"\"\n        Introduce slight shifts in the time axis of EEG signals.\n\n        Args:\n        - eeg_data (np.ndarray): Input EEG data array with shape (10000, 8)\n        - max_shift (int): Maximum allowable shift in time axis (default is 10)\n\n        Returns:\n        - shifted_eeg_data (np.ndarray): EEG data array with introduced time shifts\n        \"\"\"\n        num_samples, num_features = eeg_data.shape\n        shifted_eeg_data = np.zeros_like(eeg_data)\n\n        for i in range(num_features):\n            # Generate random shift value within the range [-max_shift, max_shift]\n            shift_amount = np.random.randint(-max_shift, max_shift + 1)\n\n            # Apply the shift to the EEG data for each feature\n            if shift_amount >= 0:\n                shifted_eeg_data[:, i] = np.concatenate((eeg_data[shift_amount:, i], np.zeros(shift_amount)))\n            else:\n                shifted_eeg_data[:, i] = np.concatenate((np.zeros(-shift_amount), eeg_data[:shift_amount, i]))\n\n        return shifted_eeg_data\n    \n    def stretch_compress_time_axis(self, eeg_data, new_length):\n        \"\"\"\n        Stretch or compress the time axis of EEG signals using linear interpolation.\n\n        Args:\n        - eeg_data (np.ndarray): Input EEG data array with shape (num_samples, num_features)\n        - new_length (int): Desired new length of the time axis\n\n        Returns:\n        - stretched_compressed_eeg_data (np.ndarray): EEG data array with stretched or compressed time axis\n        \"\"\"\n        # Determine the original length of the time axis\n        original_length = eeg_data.shape[0]\n\n        # Create new time indices based on the desired new length\n        new_indices = np.linspace(0, original_length - 1, new_length)\n\n        # Interpolate EEG data along the time axis\n        stretched_compressed_eeg_data = np.zeros((new_length, eeg_data.shape[1]))\n        for feature in range(eeg_data.shape[1]):\n            stretched_compressed_eeg_data[:, feature] = np.interp(new_indices, np.arange(original_length), eeg_data[:, feature])\n\n        return stretched_compressed_eeg_data\n    \n    def scale_eeg_amplitude(self, eeg_data, scale_range=(0.5, 1.5)):\n        \"\"\"\n        Scale the amplitude of EEG signals by a random factor.\n\n        Args:\n        - eeg_data (np.ndarray): Input EEG data array with shape (10000, 8)\n        - scale_range (tuple): Range for random scaling factor (default is (0.5, 1.5))\n\n        Returns:\n        - scaled_eeg_data (np.ndarray): Scaled EEG data array with shape (10000, 8)\n        \"\"\"\n        # Generate random scaling factors for each feature/electrode\n        scaling_factors = np.random.uniform(low=scale_range[0], high=scale_range[1], size=eeg_data.shape[1])\n\n        # Scale each feature/electrode of the EEG data by the corresponding scaling factor\n        scaled_eeg_data = eeg_data * scaling_factors\n\n        return scaled_eeg_data\n    \naug = AUG()","metadata":{"execution":{"iopub.status.busy":"2024-03-04T12:46:28.370280Z","iopub.execute_input":"2024-03-04T12:46:28.370953Z","iopub.status.idle":"2024-03-04T12:46:28.383837Z","shell.execute_reply.started":"2024-03-04T12:46:28.370891Z","shell.execute_reply":"2024-03-04T12:46:28.382445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 5: Original EEG","metadata":{}},{"cell_type":"code","source":"DATA = utilities.eeg_from_parquet(EEG_PARQUET)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T12:46:29.544417Z","iopub.execute_input":"2024-03-04T12:46:29.544823Z","iopub.status.idle":"2024-03-04T12:46:29.578734Z","shell.execute_reply.started":"2024-03-04T12:46:29.544792Z","shell.execute_reply":"2024-03-04T12:46:29.576986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 4: Augmented EEG","metadata":{}},{"cell_type":"code","source":"AUG_DATA = aug.add_noise_to_eeg(DATA, noise_level=50)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T12:46:31.786302Z","iopub.execute_input":"2024-03-04T12:46:31.787001Z","iopub.status.idle":"2024-03-04T12:46:31.795045Z","shell.execute_reply.started":"2024-03-04T12:46:31.786970Z","shell.execute_reply":"2024-03-04T12:46:31.793299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"utilities.show_eeg_two(DATA,AUG_DATA)","metadata":{"execution":{"iopub.status.busy":"2024-03-04T12:46:34.793495Z","iopub.execute_input":"2024-03-04T12:46:34.793871Z","iopub.status.idle":"2024-03-04T12:46:36.516502Z","shell.execute_reply.started":"2024-03-04T12:46:34.793842Z","shell.execute_reply":"2024-03-04T12:46:36.515218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}