{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-23T21:00:43.456430Z","iopub.execute_input":"2024-02-23T21:00:43.456700Z","iopub.status.idle":"2024-02-23T21:00:55.696654Z","shell.execute_reply.started":"2024-02-23T21:00:43.456676Z","shell.execute_reply":"2024-02-23T21:00:55.695686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install imutils","metadata":{"execution":{"iopub.status.busy":"2024-02-23T21:01:30.696683Z","iopub.execute_input":"2024-02-23T21:01:30.697043Z","iopub.status.idle":"2024-02-23T21:01:46.349555Z","shell.execute_reply.started":"2024-02-23T21:01:30.697016Z","shell.execute_reply":"2024-02-23T21:01:46.348336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, Dataset\nimport torchvision.transforms as transforms\nfrom torchvision import models\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import f1_score\nfrom sklearn.utils import shuffle\nimport cv2\nimport imutils\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport time\nfrom os import listdir\nfrom PIL import Image\nimport pyarrow.parquet as pq","metadata":{"execution":{"iopub.status.busy":"2024-02-23T21:16:54.120557Z","iopub.execute_input":"2024-02-23T21:16:54.120985Z","iopub.status.idle":"2024-02-23T21:16:54.227887Z","shell.execute_reply.started":"2024-02-23T21:16:54.120949Z","shell.execute_reply":"2024-02-23T21:16:54.227101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport pyarrow.parquet as pq\n\n# Load the labels from the CSV file\ndf = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-23T22:42:24.443353Z","iopub.execute_input":"2024-02-23T22:42:24.443883Z","iopub.status.idle":"2024-02-23T22:42:24.704866Z","shell.execute_reply.started":"2024-02-23T22:42:24.443842Z","shell.execute_reply":"2024-02-23T22:42:24.703924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2024-02-23T21:23:06.360349Z","iopub.execute_input":"2024-02-23T21:23:06.360725Z","iopub.status.idle":"2024-02-23T21:23:06.389966Z","shell.execute_reply.started":"2024-02-23T21:23:06.360696Z","shell.execute_reply":"2024-02-23T21:23:06.388841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-02-23T21:23:24.320463Z","iopub.execute_input":"2024-02-23T21:23:24.321166Z","iopub.status.idle":"2024-02-23T21:23:24.398760Z","shell.execute_reply.started":"2024-02-23T21:23:24.321133Z","shell.execute_reply":"2024-02-23T21:23:24.397768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n# Assuming df is your DataFrame\n\n# Select the columns for which you want to create boxplots\ncols_to_plot = ['eeg_id', 'eeg_sub_id', 'eeg_label_offset_seconds', 'spectrogram_id', 'spectrogram_sub_id', 'spectrogram_label_offset_seconds', 'label_id', 'patient_id', 'seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\n# Create a boxplot for each column\nfor col in cols_to_plot:\n    plt.figure(figsize=(10, 6))  # Set the figure size\n    sns.boxplot(x=df[col])  # Create the boxplot\n    plt.title(f'Boxplot of {col}')  # Set the title\n    plt.show()  # Show the plot\n","metadata":{"execution":{"iopub.status.busy":"2024-02-23T21:29:07.188316Z","iopub.execute_input":"2024-02-23T21:29:07.188719Z","iopub.status.idle":"2024-02-23T21:29:09.452087Z","shell.execute_reply.started":"2024-02-23T21:29:07.188687Z","shell.execute_reply":"2024-02-23T21:29:09.451142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"import pyarrow.parquet as pq\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\n# Define the file path\nfile_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1000913311.parquet'\n\n# Create an empty list to store the chunks\nchunks = []\n\n# Load the data in chunks\nwith pq.ParquetFile(file_path) as pq_file:\n    for i in range(pq_file.num_row_groups):\n        # Read the chunk\n        chunk = pq_file.read_row_group(i).to_pandas()\n        chunks.append(chunk)\n\n# Concatenate the chunks into a single DataFrame\ndf = pd.concat(chunks)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-23T22:27:46.697679Z","iopub.execute_input":"2024-02-23T22:27:46.698366Z","iopub.status.idle":"2024-02-23T22:27:46.731694Z","shell.execute_reply.started":"2024-02-23T22:27:46.698330Z","shell.execute_reply":"2024-02-23T22:27:46.730778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Define the file path\nfile_path = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/1000913311.parquet'\n\n# Load the data\ndf = pd.read_parquet(file_path)\n\n# Remove leading and trailing spaces from column names\ndf.columns = df.columns.str.strip()\n\n# Create a heatmap\nplt.figure(figsize=(12, 8))\nsns.heatmap(df.corr(), annot=True, cmap='coolwarm', fmt='.2f')\nplt.title('Correlation Heatmap of EEG Signals')\nplt.show()\n\n# Create subplots for each electrode\nfig, axes = plt.subplots(nrows=5, ncols=4, figsize=(20, 20))\nfor i, col in enumerate(df.columns):\n    ax = axes[i // 4, i % 4]\n    ax.plot(df.index, df[col])\n    ax.set_title(col)\n    ax.set_xlabel('Timestamp')\n    ax.set_ylabel('EEG Signal')\nplt.tight_layout()\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-02-23T22:42:32.513234Z","iopub.execute_input":"2024-02-23T22:42:32.514126Z","iopub.status.idle":"2024-02-23T22:42:40.359600Z","shell.execute_reply.started":"2024-02-23T22:42:32.514093Z","shell.execute_reply":"2024-02-23T22:42:40.358597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}}]}