{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":160160108,"sourceType":"kernelVersion"},{"sourceId":160620406,"sourceType":"kernelVersion"},{"sourceId":6120,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":4603,"modelId":2797}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix\nimport tensorflow as tf\nimport os\nimport torch","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:10:26.936427Z","iopub.execute_input":"2025-03-24T15:10:26.936695Z","iopub.status.idle":"2025-03-24T15:10:43.010942Z","shell.execute_reply.started":"2025-03-24T15:10:26.936666Z","shell.execute_reply":"2025-03-24T15:10:43.010001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\n\ngpus = tf.config.list_physical_devices('GPU')\nif len(gpus)<=1: \n    strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n    print(f'Using {len(gpus)} GPU')\nelse: \n    strategy = tf.distribute.MirroredStrategy()\n    print(f'Using {len(gpus)} GPUs')\n    \nMIX = True\nif MIX:\n    tf.config.optimizer.set_experimental_options({\"auto_mixed_precision\": True})\n    print('Mixed precision enabled')\nelse:\n    print('Using full precision')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:11:02.000586Z","iopub.execute_input":"2025-03-24T15:11:02.000947Z","iopub.status.idle":"2025-03-24T15:11:02.379970Z","shell.execute_reply.started":"2025-03-24T15:11:02.000903Z","shell.execute_reply":"2025-03-24T15:11:02.378838Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission_df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv')\nsample_submission_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:11:15.967356Z","iopub.execute_input":"2025-03-24T15:11:15.967657Z","iopub.status.idle":"2025-03-24T15:11:16.011581Z","shell.execute_reply.started":"2025-03-24T15:11:15.967635Z","shell.execute_reply":"2025-03-24T15:11:16.010820Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_submission_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:11:28.735666Z","iopub.execute_input":"2025-03-24T15:11:28.736005Z","iopub.status.idle":"2025-03-24T15:11:28.757693Z","shell.execute_reply.started":"2025-03-24T15:11:28.735982Z","shell.execute_reply":"2025-03-24T15:11:28.757054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\ntrain_data.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:11:39.800332Z","iopub.execute_input":"2025-03-24T15:11:39.800676Z","iopub.status.idle":"2025-03-24T15:11:40.086536Z","shell.execute_reply.started":"2025-03-24T15:11:39.800649Z","shell.execute_reply":"2025-03-24T15:11:40.085427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.head(20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:11:56.425169Z","iopub.execute_input":"2025-03-24T15:11:56.425482Z","iopub.status.idle":"2025-03-24T15:11:56.442174Z","shell.execute_reply.started":"2025-03-24T15:11:56.425454Z","shell.execute_reply":"2025-03-24T15:11:56.441395Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sns.countplot(x='expert_consensus', data=train_data)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:12:08.331508Z","iopub.execute_input":"2025-03-24T15:12:08.331830Z","iopub.status.idle":"2025-03-24T15:12:08.597288Z","shell.execute_reply.started":"2025-03-24T15:12:08.331808Z","shell.execute_reply":"2025-03-24T15:12:08.596434Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class_names = ['seizure', 'lpd', 'gpd', 'lrda', 'grda', 'other']\nclass_name_to_index = {'Seizure' : 0 , 'LPD' : 1 , \n                       'LRDA' : 3 , 'GPD' : 2 , \n                       'GRDA' : 4 , 'Other' : 5}\n\nplt.figure(figsize=(15, 10)) \n\nfor i, class_name in enumerate(class_names):\n    plt.subplot(2, 3, i+1) \n    sns.countplot(x=f'{class_name}_vote', data=train_data)\n    plt.title(f'Distribution of {class_name} votes')\n    plt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:12:28.071888Z","iopub.execute_input":"2025-03-24T15:12:28.072183Z","iopub.status.idle":"2025-03-24T15:12:30.203545Z","shell.execute_reply.started":"2025-03-24T15:12:28.072163Z","shell.execute_reply":"2025-03-24T15:12:30.202627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data.hist(bins=10, figsize=(15, 20), layout=(7, 2))\nplt.suptitle('Feature Distributions')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:12:41.467851Z","iopub.execute_input":"2025-03-24T15:12:41.468204Z","iopub.status.idle":"2025-03-24T15:12:43.555267Z","shell.execute_reply.started":"2025-03-24T15:12:41.468180Z","shell.execute_reply":"2025-03-24T15:12:43.554366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"vote_columns = [f'{name}_vote' for name in class_names]\ncorr_matrix = train_data[vote_columns].corr()\n\nplt.figure(figsize=(12, 8))\nsns.heatmap(corr_matrix, annot=True, fmt='.2f', cmap='viridis')\nplt.title('Correlation Matrix for Vote Columns')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:12:57.490136Z","iopub.execute_input":"2025-03-24T15:12:57.490459Z","iopub.status.idle":"2025-03-24T15:12:57.796282Z","shell.execute_reply.started":"2025-03-24T15:12:57.490433Z","shell.execute_reply":"2025-03-24T15:12:57.795356Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15, 10))\n\nfor i, col in enumerate([f'{name}_vote' for name in class_names]):\n    plt.subplot(2, 3, i+1)\n    sns.boxplot(y='expert_consensus', x=col, data=train_data)\n    plt.title(f'Box Plot of {col} vs Expert Consensus')\n    plt.tight_layout() \n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:13:34.293891Z","iopub.execute_input":"2025-03-24T15:13:34.294220Z","iopub.status.idle":"2025-03-24T15:13:36.032833Z","shell.execute_reply.started":"2025-03-24T15:13:34.294197Z","shell.execute_reply":"2025-03-24T15:13:36.032011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eeg_dir = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs'\nspectrogram_dir = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms'\nmetadata_path = '/kaggle/input/hms-harmful-brain-activity-classification/train.csv'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:13:54.391079Z","iopub.execute_input":"2025-03-24T15:13:54.391375Z","iopub.status.idle":"2025-03-24T15:13:54.394977Z","shell.execute_reply.started":"2025-03-24T15:13:54.391354Z","shell.execute_reply":"2025-03-24T15:13:54.394178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_data(ids, file_dir):\n    file_path = f\"{file_dir}/{int(ids)}.parquet\"\n    data_df = pd.read_parquet(file_path)\n    return data_df\n\ndef load_eeg_data(ids):\n    return load_data(ids, eeg_dir)\n\ndef load_spectrogram_data(ids):\n    return load_data(ids, spectrogram_dir).drop(columns=['time'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:14:02.638214Z","iopub.execute_input":"2025-03-24T15:14:02.638491Z","iopub.status.idle":"2025-03-24T15:14:02.642932Z","shell.execute_reply.started":"2025-03-24T15:14:02.638469Z","shell.execute_reply":"2025-03-24T15:14:02.641971Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_eeg_example = load_eeg_data(1628180742)\ndf_eeg_example.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:14:17.482030Z","iopub.execute_input":"2025-03-24T15:14:17.482332Z","iopub.status.idle":"2025-03-24T15:14:17.683123Z","shell.execute_reply.started":"2025-03-24T15:14:17.482310Z","shell.execute_reply":"2025-03-24T15:14:17.682360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_spectro_example = load_spectrogram_data(999431)\ndf_spectro_example.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:14:29.242771Z","iopub.execute_input":"2025-03-24T15:14:29.243077Z","iopub.status.idle":"2025-03-24T15:14:29.315527Z","shell.execute_reply.started":"2025-03-24T15:14:29.243056Z","shell.execute_reply":"2025-03-24T15:14:29.314728Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_spectro_example.columns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:14:38.186195Z","iopub.execute_input":"2025-03-24T15:14:38.186474Z","iopub.status.idle":"2025-03-24T15:14:38.191709Z","shell.execute_reply.started":"2025-03-24T15:14:38.186454Z","shell.execute_reply":"2025-03-24T15:14:38.191040Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"load_eeg_data(train_data['eeg_id'][190])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:16:32.223602Z","iopub.execute_input":"2025-03-24T15:16:32.223986Z","iopub.status.idle":"2025-03-24T15:16:32.286136Z","shell.execute_reply.started":"2025-03-24T15:16:32.223956Z","shell.execute_reply":"2025-03-24T15:16:32.285268Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"load_spectrogram_data(train_data['spectrogram_id'][28])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:16:45.733051Z","iopub.execute_input":"2025-03-24T15:16:45.733359Z","iopub.status.idle":"2025-03-24T15:16:45.852464Z","shell.execute_reply.started":"2025-03-24T15:16:45.733335Z","shell.execute_reply":"2025-03-24T15:16:45.851575Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = train_data.drop(columns=['eeg_sub_id','eeg_label_offset_seconds',\n                         'spectrogram_sub_id','spectrogram_label_offset_seconds',\n                         'label_id','patient_id'])\n\ndf_train = df_train.drop_duplicates().reset_index()\ndf_train.drop(columns=['index'], inplace=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:16:55.328853Z","iopub.execute_input":"2025-03-24T15:16:55.329169Z","iopub.status.idle":"2025-03-24T15:16:55.365598Z","shell.execute_reply.started":"2025-03-24T15:16:55.329147Z","shell.execute_reply":"2025-03-24T15:16:55.364951Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train['total'] = df_train[vote_columns].sum(axis=1)\ndf_train[vote_columns] = df_train[vote_columns].div(df_train['total'], axis=0)\ndf_train.drop(columns=['total'], inplace=True)\n\ndf_train['expert_consensus'] = df_train['expert_consensus'].map(class_name_to_index)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:17:06.763198Z","iopub.execute_input":"2025-03-24T15:17:06.763471Z","iopub.status.idle":"2025-03-24T15:17:06.784304Z","shell.execute_reply.started":"2025-03-24T15:17:06.763451Z","shell.execute_reply":"2025-03-24T15:17:06.783451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:17:13.864977Z","iopub.execute_input":"2025-03-24T15:17:13.865292Z","iopub.status.idle":"2025-03-24T15:17:13.878458Z","shell.execute_reply.started":"2025-03-24T15:17:13.865268Z","shell.execute_reply":"2025-03-24T15:17:13.877591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train[vote_columns]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:17:22.435361Z","iopub.execute_input":"2025-03-24T15:17:22.435659Z","iopub.status.idle":"2025-03-24T15:17:22.449632Z","shell.execute_reply.started":"2025-03-24T15:17:22.435636Z","shell.execute_reply":"2025-03-24T15:17:22.448862Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def preprocess(dataframe, eeg_dir, spectrogram_dir, vote_columns):\n    eeg_features_list = []\n    spectrogram_features_list = []\n    labels_list = []\n\n    for idx in range(len(dataframe)):\n        eeg_id = dataframe.iloc[idx]['eeg_id']\n        spectrogram_id = dataframe.iloc[idx]['spectrogram_id']\n\n        eeg_data = load_data(eeg_id, eeg_dir)\n        eeg_features = extract_features(eeg_data)\n        eeg_features_list.append(eeg_features)\n\n        spectrogram_data = load_data(spectrogram_id, spectrogram_dir).drop(columns=['time'])\n        spectrogram_features = extract_features(spectrogram_data)\n        spectrogram_features_list.append(spectrogram_features)\n\n        label = dataframe.iloc[idx][vote_columns].values\n        labels_list.append(label)\n\n    eeg_features_tensor = torch.tensor(eeg_features_list, dtype=torch.float32)\n    spectrogram_features_tensor = torch.tensor(spectrogram_features_list, dtype=torch.float32)\n    labels_tensor = torch.tensor(labels_list, dtype=torch.float32)\n\n    return eeg_features_tensor, spectrogram_features_tensor, labels_tensor\n\n\ndef extract_features(df):\n    current_size = len(df)\n\n    # Basic statistical features\n    min_values = df.min()\n    max_values = df.max()\n    mean_values = df.mean()\n    std_values = df.std()\n\n    # Time-domain features\n    rms_values = np.sqrt(np.mean(np.square(df), axis=0))\n    var_values = df.var()\n    skew_values = df.skew()\n    kurtosis_values = df.kurtosis()\n\n    # Concatenate all features\n    features = np.concatenate([\n        min_values, max_values, mean_values, std_values, \n        rms_values, var_values, skew_values, kurtosis_values\n    ])\n\n\n    return features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:17:36.989160Z","iopub.execute_input":"2025-03-24T15:17:36.989489Z","iopub.status.idle":"2025-03-24T15:17:36.996748Z","shell.execute_reply.started":"2025-03-24T15:17:36.989463Z","shell.execute_reply":"2025-03-24T15:17:36.995729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"eeg_features_tensor, spectrogram_features_tensor, labels_tensor = preprocess(df_train, eeg_dir, spectrogram_dir, vote_columns)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T15:18:09.117725Z","iopub.execute_input":"2025-03-24T15:18:09.118073Z","iopub.status.idle":"2025-03-24T15:52:01.653126Z","shell.execute_reply.started":"2025-03-24T15:18:09.118042Z","shell.execute_reply":"2025-03-24T15:52:01.651784Z"}},"outputs":[],"execution_count":null}]}