{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-17T07:22:33.361296Z","iopub.execute_input":"2024-01-17T07:22:33.362490Z","iopub.status.idle":"2024-01-17T07:22:48.375697Z","shell.execute_reply.started":"2024-01-17T07:22:33.362450Z","shell.execute_reply":"2024-01-17T07:22:48.374118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nsns.set(style=\"whitegrid\")\ntrain = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\n\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:22:48.377780Z","iopub.execute_input":"2024-01-17T07:22:48.378330Z","iopub.status.idle":"2024-01-17T07:22:49.373627Z","shell.execute_reply.started":"2024-01-17T07:22:48.378294Z","shell.execute_reply":"2024-01-17T07:22:49.372533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rows, columns= train.shape\nrows,columns","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:23.243694Z","iopub.execute_input":"2024-01-17T07:23:23.244080Z","iopub.status.idle":"2024-01-17T07:23:23.251154Z","shell.execute_reply.started":"2024-01-17T07:23:23.244051Z","shell.execute_reply":"2024-01-17T07:23:23.250105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:23.656871Z","iopub.execute_input":"2024-01-17T07:23:23.657791Z","iopub.status.idle":"2024-01-17T07:23:23.712169Z","shell.execute_reply.started":"2024-01-17T07:23:23.657739Z","shell.execute_reply":"2024-01-17T07:23:23.711039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:23.961301Z","iopub.execute_input":"2024-01-17T07:23:23.962196Z","iopub.status.idle":"2024-01-17T07:23:24.091494Z","shell.execute_reply.started":"2024-01-17T07:23:23.962146Z","shell.execute_reply":"2024-01-17T07:23:24.090507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_columns = train.select_dtypes(include=['object', 'category']).columns\ncategorical_summary = train[categorical_columns].describe()\ncategorical_summary","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:24.166422Z","iopub.execute_input":"2024-01-17T07:23:24.167240Z","iopub.status.idle":"2024-01-17T07:23:24.215336Z","shell.execute_reply.started":"2024-01-17T07:23:24.167192Z","shell.execute_reply":"2024-01-17T07:23:24.214253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list(set(train['expert_consensus'].unique()))","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:24.286668Z","iopub.execute_input":"2024-01-17T07:23:24.287455Z","iopub.status.idle":"2024-01-17T07:23:24.303887Z","shell.execute_reply.started":"2024-01-17T07:23:24.287412Z","shell.execute_reply":"2024-01-17T07:23:24.302794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.countplot(data=train, x='expert_consensus')\nplt.title('Distribution of Expert Consensus')\nplt.xlabel('Expert Consensus')\nplt.ylabel('Count')\nplt.xticks(rotation=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:24.474284Z","iopub.execute_input":"2024-01-17T07:23:24.475122Z","iopub.status.idle":"2024-01-17T07:23:25.051575Z","shell.execute_reply.started":"2024-01-17T07:23:24.475071Z","shell.execute_reply":"2024-01-17T07:23:25.050450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 6))\nsns.histplot(train['patient_id'], bins=30, kde=False)\nplt.title('Distribution of Patient ID')\nplt.xlabel('Patient ID')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:25.053462Z","iopub.execute_input":"2024-01-17T07:23:25.053841Z","iopub.status.idle":"2024-01-17T07:23:25.506951Z","shell.execute_reply.started":"2024-01-17T07:23:25.053809Z","shell.execute_reply":"2024-01-17T07:23:25.505758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"targets = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\nplt.figure(figsize=(15, 10))\nfor i, column in enumerate(targets, 1):\n    plt.subplot(2, 4, i)\n    sns.histplot(train[column], kde=False, bins=30)\n    plt.title(column)\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:25.508466Z","iopub.execute_input":"2024-01-17T07:23:25.508832Z","iopub.status.idle":"2024-01-17T07:23:28.595786Z","shell.execute_reply.started":"2024-01-17T07:23:25.508799Z","shell.execute_reply":"2024-01-17T07:23:28.594633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correlation_targets = train[targets].corr()\nplt.figure(figsize=(12, 8))\nsns.heatmap(correlation_targets, annot=True, cmap='coolwarm', fmt=\".2f\")\nplt.title('Correlation Matrix of Vote Columns')\nplt.show()\n\nplt.figure(figsize=(12, 10))\nfor i, column in enumerate(targets, 1):\n    plt.subplot(3, 2, i)\n    sns.violinplot(data=train, x='expert_consensus', y=column)\n    plt.title(f'Distribution of {column} by Expert Consensus')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:28.598508Z","iopub.execute_input":"2024-01-17T07:23:28.598877Z","iopub.status.idle":"2024-01-17T07:23:35.219020Z","shell.execute_reply.started":"2024-01-17T07:23:28.598847Z","shell.execute_reply":"2024-01-17T07:23:35.217725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"offset_stats = train[['eeg_label_offset_seconds', 'spectrogram_label_offset_seconds']].describe()\n\nplt.figure(figsize=(12, 6))\nsns.histplot(train['eeg_label_offset_seconds'], bins=30, kde=True)\nplt.title('Distribution of EEG Label Offset Seconds')\nplt.xlabel('EEG Label Offset Seconds')\nplt.ylabel('Count')\nplt.show()\n\nplt.figure(figsize=(12, 6))\nsns.histplot(train['spectrogram_label_offset_seconds'], bins=30, kde=True)\nplt.title('Distribution of Spectrogram Label Offset Seconds')\nplt.xlabel('Spectrogram Label Offset Seconds')\nplt.ylabel('Count')\nplt.show()\n\noffset_stats","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:35.220735Z","iopub.execute_input":"2024-01-17T07:23:35.221552Z","iopub.status.idle":"2024-01-17T07:23:37.134838Z","shell.execute_reply.started":"2024-01-17T07:23:35.221507Z","shell.execute_reply":"2024-01-17T07:23:37.133737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_eegs = len(train['eeg_id'].unique())\ntotal_eegs","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:37.136355Z","iopub.execute_input":"2024-01-17T07:23:37.136743Z","iopub.status.idle":"2024-01-17T07:23:37.147562Z","shell.execute_reply.started":"2024-01-17T07:23:37.136713Z","shell.execute_reply":"2024-01-17T07:23:37.146548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_eeg_label_offset_seconds = sorted(list(train['eeg_label_offset_seconds'].unique()))\nlen(all_eeg_label_offset_seconds), str(all_eeg_label_offset_seconds[0:5]), str(all_eeg_label_offset_seconds[-5:])","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:37.148627Z","iopub.execute_input":"2024-01-17T07:23:37.149015Z","iopub.status.idle":"2024-01-17T07:23:37.159228Z","shell.execute_reply.started":"2024-01-17T07:23:37.148984Z","shell.execute_reply":"2024-01-17T07:23:37.158122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_spectrogram_label_offset_seconds = sorted(list(train['spectrogram_label_offset_seconds'].unique()))\nlen(all_spectrogram_label_offset_seconds), str(all_spectrogram_label_offset_seconds[0:5]), str(all_spectrogram_label_offset_seconds[-5:])","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:37.161171Z","iopub.execute_input":"2024-01-17T07:23:37.162061Z","iopub.status.idle":"2024-01-17T07:23:37.175611Z","shell.execute_reply.started":"2024-01-17T07:23:37.162014Z","shell.execute_reply":"2024-01-17T07:23:37.174235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eeg_sub_id_count_per_eeg_id = train.groupby('eeg_id')['eeg_sub_id'].nunique()\nspectrogram_sub_id_count_per_spectrogram_id = train.groupby('spectrogram_id')['spectrogram_sub_id'].nunique()\n\nplt.figure(figsize=(12, 6))\nsns.histplot(eeg_sub_id_count_per_eeg_id, bins=50, kde=True)\nplt.title('EEG Sub-ID Count per EEG ID')\nplt.xlabel('Count of EEG Sub-ID per EEG ID')\nplt.ylabel('Frequency')\nplt.show()\n\nplt.figure(figsize=(12, 6))\nsns.histplot(spectrogram_sub_id_count_per_spectrogram_id, bins=50, kde=True)\nplt.title('Spectrogram Sub-ID Count per Spectrogram ID')\nplt.xlabel('Count of Spectrogram Sub-ID per Spectrogram ID')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:37.177298Z","iopub.execute_input":"2024-01-17T07:23:37.177628Z","iopub.status.idle":"2024-01-17T07:23:38.369455Z","shell.execute_reply.started":"2024-01-17T07:23:37.177600Z","shell.execute_reply":"2024-01-17T07:23:38.368173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vote_counts_by_consensus = train.groupby('expert_consensus')[targets].sum()\n\nplt.figure(figsize=(12, 8))\nvote_counts_by_consensus.plot(kind='bar', stacked=True)\nplt.title('Overall Vote Counts by Expert Consensus')\nplt.xlabel('Expert Consensus')\nplt.ylabel('Total Votes')\nplt.xticks(rotation=45)\nplt.legend(title='Vote Types')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:38.373174Z","iopub.execute_input":"2024-01-17T07:23:38.373554Z","iopub.status.idle":"2024-01-17T07:23:38.856795Z","shell.execute_reply.started":"2024-01-17T07:23:38.373522Z","shell.execute_reply":"2024-01-17T07:23:38.855613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cumulative_votes = train.groupby('eeg_label_offset_seconds')[targets].sum().cumsum().reset_index()\n\nplt.figure(figsize=(12, 8))\nfor column in targets:\n    plt.plot(cumulative_votes['eeg_label_offset_seconds'], cumulative_votes[column], label=column)\n\nplt.title('Vote Counts Over EEG Label Offset Seconds')\nplt.xlabel('EEG Label Offset Seconds')\nplt.ylabel('Total Votes')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:38.858463Z","iopub.execute_input":"2024-01-17T07:23:38.858941Z","iopub.status.idle":"2024-01-17T07:23:39.384008Z","shell.execute_reply.started":"2024-01-17T07:23:38.858894Z","shell.execute_reply":"2024-01-17T07:23:39.382815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cumulative_votes = train.groupby('spectrogram_label_offset_seconds')[targets].sum().cumsum().reset_index()\n\nplt.figure(figsize=(12, 8))\nfor column in targets:\n    plt.plot(cumulative_votes['spectrogram_label_offset_seconds'], cumulative_votes[column], label=column)\n\nplt.title('Vote Counts Over Spectrogram Offset Seconds')\nplt.xlabel('EEG Label Offset Seconds')\nplt.ylabel('Total Votes')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:39.385534Z","iopub.execute_input":"2024-01-17T07:23:39.386020Z","iopub.status.idle":"2024-01-17T07:23:40.107504Z","shell.execute_reply.started":"2024-01-17T07:23:39.385980Z","shell.execute_reply":"2024-01-17T07:23:40.106336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cumulative_votes","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:40.109169Z","iopub.execute_input":"2024-01-17T07:23:40.109486Z","iopub.status.idle":"2024-01-17T07:23:40.128100Z","shell.execute_reply.started":"2024-01-17T07:23:40.109459Z","shell.execute_reply":"2024-01-17T07:23:40.126595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted_data = train.sort_values(by=['eeg_id', 'eeg_sub_id'])\n\nsorted_data['offset_difference'] = sorted_data.groupby('eeg_id')['eeg_label_offset_seconds'].diff()\n\noffset_differences = sorted_data['offset_difference'].dropna()\n\noffset_difference_stats = offset_differences.describe()\n\nplt.figure(figsize=(12, 6))\nsns.histplot(offset_differences, bins=30, kde=True)\nplt.title('Offset Differences within EEG IDs')\nplt.xlabel('Offset Difference (Seconds)')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:40.129691Z","iopub.execute_input":"2024-01-17T07:23:40.130186Z","iopub.status.idle":"2024-01-17T07:23:41.102339Z","shell.execute_reply.started":"2024-01-17T07:23:40.130144Z","shell.execute_reply":"2024-01-17T07:23:41.101364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted_data = train.sort_values(by=['spectrogram_id', 'spectrogram_sub_id'])\n\nsorted_data['offset_difference'] = sorted_data.groupby('spectrogram_id')['spectrogram_label_offset_seconds'].diff()\n\noffset_differences = sorted_data['offset_difference'].dropna()\n\noffset_difference_stats = offset_differences.describe()\n\nplt.figure(figsize=(12, 6))\nsns.histplot(offset_differences, bins=30, kde=True)\nplt.title('Offset Differences within Spectrogram IDs')\nplt.xlabel('Offset Difference (Seconds)')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:41.103704Z","iopub.execute_input":"2024-01-17T07:23:41.104042Z","iopub.status.idle":"2024-01-17T07:23:42.076888Z","shell.execute_reply.started":"2024-01-17T07:23:41.104012Z","shell.execute_reply":"2024-01-17T07:23:42.075706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_patients = train['patient_id'].sample(20, random_state=1).values\nsample_data = train[train['patient_id'].isin(sample_patients)]\n\nfor i, vote_type in enumerate(targets, 1):\n    plt.figure(figsize=(15, 10))\n    sns.boxplot(x='patient_id', y=vote_type, data=sample_data)\n    plt.title(f'Distribution of {vote_type} for Selected Patients')\n    plt.xlabel('Patient ID')\n    plt.ylabel(f'{vote_type} Count')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:42.078493Z","iopub.execute_input":"2024-01-17T07:23:42.079281Z","iopub.status.idle":"2024-01-17T07:23:46.301901Z","shell.execute_reply.started":"2024-01-17T07:23:42.079235Z","shell.execute_reply":"2024-01-17T07:23:46.300721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15, 10))\n\n\nfor i, patient_id in enumerate(sample_patients, 1):\n    plt.figure(figsize=(15, 10))\n    patient_data = train[train['patient_id'] == patient_id]\n    correlation_matrix = patient_data[targets].corr()\n    sns.heatmap(correlation_matrix, annot=True, cmap='coolwarm', fmt=\".2f\")\n    plt.title(f'Correlation of Votes for Patient ID {patient_id}')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:46.303478Z","iopub.execute_input":"2024-01-17T07:23:46.303863Z","iopub.status.idle":"2024-01-17T07:23:56.526426Z","shell.execute_reply.started":"2024-01-17T07:23:46.303831Z","shell.execute_reply":"2024-01-17T07:23:56.523749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_votes_per_pat = train.groupby('patient_id')[targets].sum().sum(axis=1)\nnormalized_votes = train.groupby('patient_id')[targets].sum().div(total_votes_per_pat, axis=0)\nmean_vote_ratio = normalized_votes.mean()\nprint( mean_vote_ratio )","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:56.527828Z","iopub.execute_input":"2024-01-17T07:23:56.528187Z","iopub.status.idle":"2024-01-17T07:23:56.554071Z","shell.execute_reply.started":"2024-01-17T07:23:56.528153Z","shell.execute_reply":"2024-01-17T07:23:56.552710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gap = 1 - sum([round(v,6) for _, v in mean_vote_ratio.items()])\nprint(gap)\nmean_vote_ratio['other_vote'] += gap","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:56.555590Z","iopub.execute_input":"2024-01-17T07:23:56.556351Z","iopub.status.idle":"2024-01-17T07:23:56.563730Z","shell.execute_reply.started":"2024-01-17T07:23:56.556313Z","shell.execute_reply":"2024-01-17T07:23:56.562425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum([round(v,5) for _, v in mean_vote_ratio.items()])","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:56.564846Z","iopub.execute_input":"2024-01-17T07:23:56.565166Z","iopub.status.idle":"2024-01-17T07:23:56.575130Z","shell.execute_reply.started":"2024-01-17T07:23:56.565139Z","shell.execute_reply":"2024-01-17T07:23:56.574058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_vote_ratio","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:56.576069Z","iopub.execute_input":"2024-01-17T07:23:56.576386Z","iopub.status.idle":"2024-01-17T07:23:56.587982Z","shell.execute_reply.started":"2024-01-17T07:23:56.576359Z","shell.execute_reply":"2024-01-17T07:23:56.586589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv\")\nfor target in targets:\n    sub[target] = mean_vote_ratio[target]\nsub","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:23:56.589436Z","iopub.execute_input":"2024-01-17T07:23:56.589914Z","iopub.status.idle":"2024-01-17T07:23:56.612533Z","shell.execute_reply.started":"2024-01-17T07:23:56.589872Z","shell.execute_reply":"2024-01-17T07:23:56.611425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-17T07:24:29.193900Z","iopub.execute_input":"2024-01-17T07:24:29.194387Z","iopub.status.idle":"2024-01-17T07:24:29.205695Z","shell.execute_reply.started":"2024-01-17T07:24:29.194349Z","shell.execute_reply":"2024-01-17T07:24:29.204406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}