{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport numpy as np\nfrom tqdm.notebook import tqdm\n\nsns.set(style=\"whitegrid\")\ntrain = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/test.csv\")\nall_df = pd.concat([train, test]).reset_index(drop=True)\n\ndisplay(train.head())\ndisplay(test.head())\ndisplay(all_df.head())","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-29T01:48:33.203227Z","iopub.execute_input":"2024-01-29T01:48:33.204029Z","iopub.status.idle":"2024-01-29T01:48:33.412144Z","shell.execute_reply.started":"2024-01-29T01:48:33.203983Z","shell.execute_reply":"2024-01-29T01:48:33.411428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train.eeg_id.unique()))\nprint(len(train.spectrogram_id.unique()))\nprint(len(train.patient_id.unique()))\nprint(len(train.eeg_id.unique())/len(train.patient_id.unique()))\nprint(len(train.spectrogram_id.unique())/len(train.patient_id.unique()))","metadata":{"execution":{"iopub.status.busy":"2024-01-28T07:08:09.586349Z","iopub.execute_input":"2024-01-28T07:08:09.586722Z","iopub.status.idle":"2024-01-28T07:08:09.605725Z","shell.execute_reply.started":"2024-01-28T07:08:09.586688Z","shell.execute_reply":"2024-01-28T07:08:09.604239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_spectrogram_dir = '/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/'\ntrain_spectrogram_files = os.listdir(train_spectrogram_dir)\nprint(f'There are {len(train_spectrogram_files)} train spectrogram parquets')\ntest_spectrogram_dir = '/kaggle/input/hms-harmful-brain-activity-classification/test_spectrograms/'\ntest_spectrogram_files = os.listdir(test_spectrogram_dir)\nprint(f'There are {len(test_spectrogram_files)} test spectrogram parquets')\ntrain_eeg_dir = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/'\ntrain_eeg_files = os.listdir(train_eeg_dir)\nprint(f'There are {len(train_eeg_files)} train eeg parquets')\ntest_eeg_dir = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs/'\ntest_eeg_files = os.listdir(test_eeg_dir)\nprint(f'There are {len(test_eeg_files)} test eeg parquets')","metadata":{"execution":{"iopub.status.busy":"2024-01-28T07:24:11.281438Z","iopub.execute_input":"2024-01-28T07:24:11.282258Z","iopub.status.idle":"2024-01-28T07:24:11.306324Z","shell.execute_reply.started":"2024-01-28T07:24:11.282218Z","shell.execute_reply":"2024-01-28T07:24:11.304890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_files_info(files, file_dir):\n    nan_ratio = []\n    shapes = []\n    for file in tqdm(files):    \n        data = np.array(pd.read_parquet(f\"{file_dir}{file}\"))\n        nan_ratio.append(np.isnan(data).sum() / len(data.flatten()))\n        shapes.append(data.shape)\n    nan_ratio = np.array(nan_ratio)\n    shapes = np.array(shapes)\n    return nan_ratio, shapes","metadata":{"execution":{"iopub.status.busy":"2024-01-28T07:33:24.508642Z","iopub.execute_input":"2024-01-28T07:33:24.509071Z","iopub.status.idle":"2024-01-28T07:33:24.516572Z","shell.execute_reply.started":"2024-01-28T07:33:24.509035Z","shell.execute_reply":"2024-01-28T07:33:24.515290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_spectrogram_nan_ratio, train_spectrogram_shapes = get_files_info(train_spectrogram_files, train_spectrogram_dir)\ntest_spectrogram_nan_ratio, test_spectrogram_shapes = get_files_info(test_spectrogram_files, test_spectrogram_dir)\n# train_eeg_nan_ratio, train_eeg_shapes = get_files_info(train_eeg_files, train_eeg_dir)\ntest_eeg_nan_ratio, test_eeg_shapes = get_files_info(test_eeg_files, test_eeg_dir)\n\n# print(train_spectrogram_nan_ratio.mean())\nprint(test_spectrogram_nan_ratio.mean())\n# print(train_eeg_nan_ratio.mean())\nprint(test_eeg_nan_ratio.mean())\n# print(np.unique(train_spectrogram_shapes))\nprint(np.unique(test_spectrogram_shapes))\n# print(np.unique(train_eeg_shapes))\nprint(np.unique(test_eeg_shapes))\n\n# \n# \n# 0.02423864616555035\n# 0.0\n# 0.0016409762644298224\n# 0.0","metadata":{"execution":{"iopub.status.busy":"2024-01-28T07:36:52.168209Z","iopub.execute_input":"2024-01-28T07:36:52.168794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hypothesis0  = False\nhypothesis1  = False\n\nif test_spectrogram_nan_ratio.mean()<0.025:\n    hypothesis0 = True    \n    hypothesis1  = True\nif test_spectrogram_nan_ratio.mean()<0.02:\n    hypothesis0 = True    \n    hypothesis1  = False\nif test_spectrogram_nan_ratio.mean()<0.015:\n    hypothesis0 = False    \n    hypothesis1  = True\n    \nprint(f'hypothesis0: {hypothesis0}')\nprint(f'hypothesis1: {hypothesis1}')","metadata":{"execution":{"iopub.status.busy":"2024-01-29T01:51:19.049943Z","iopub.execute_input":"2024-01-29T01:51:19.050309Z","iopub.status.idle":"2024-01-29T01:51:19.072171Z","shell.execute_reply.started":"2024-01-29T01:51:19.050280Z","shell.execute_reply":"2024-01-29T01:51:19.071064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nhypotheses=[]\n#there are no overlap of eeg ids in test set\nhypotheses.append(len(test.eeg_id.unique()) == len(test))\n#there are no overlap of spectrogram ids in test set\nhypotheses.append(len(test.spectrogram_id.unique()) == len(test))\n#there are overlap of patient ids in test set\nhypotheses.append(len(test.patient_id.unique()) != len(test))\n#eeg ids in test csv is the same as those in test_eeg dir\nhypotheses.append(len(test_eeg_files) == len(test))\n#spectrogram ids in test csv is the same as those ids in test_spectrogram dir\nhypotheses.append(len(test_spectrogram_files) == len(test))\n#number of spectrogram per patient < 6\nhypotheses.append(len(train.spectrogram_id.unique())/len(train.patient_id.unique()) < 6)\n#number of spectrogram per patient > 5\nhypotheses.append(len(train.spectrogram_id.unique())/len(train.patient_id.unique()) > 5)\n#eeg_ids of train and test set do not overlap\nhypotheses.append(len(all_df.eeg_id.unique())==(len(train.eeg_id.unique())+len(test.eeg_id.unique())))\n#spectrogram_ids of train and test set do not overlap\nhypotheses.append(len(all_df.spectrogram_id.unique())==(len(train.spectrogram_id.unique())+len(test.spectrogram_id.unique())))\n#patient_ids of train and test set do not overlap\nhypotheses.append(len(all_df.patient_id.unique())==(len(train.patient_id.unique())+len(test.patient_id.unique())))\n#shape of eeg is [20,10000]\nhypotheses.append(np.array_equal(np.unique(test_eeg_shapes),np.array([20,10000])))\n#shape of spectrogram is [300, 401]\nhypotheses.append(np.array_equal(np.unique(test_spectrogram_shapes),np.array([300, 401])))\n#num nan/data in test spectrogram < 0.03\nhypotheses.append(test_spectrogram_nan_ratio.mean()<0.03)\n#num nan/data in test spectrogram < 0.01\nhypotheses.append(test_spectrogram_nan_ratio.mean()>0.01)\n#there are no nan in eeg data\nhypotheses.append(test_eeg_nan_ratio.mean()==0)\n\nprint(f'hypotheses: {hypotheses}')\nhypotheses = all(hypotheses)\nprint(f'hypotheses: {hypotheses}')","metadata":{"execution":{"iopub.status.busy":"2024-01-28T07:11:11.638192Z","iopub.execute_input":"2024-01-28T07:11:11.638670Z","iopub.status.idle":"2024-01-28T07:11:11.648168Z","shell.execute_reply.started":"2024-01-28T07:11:11.638632Z","shell.execute_reply":"2024-01-28T07:11:11.646420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv\")\n\n\nif (hypothesis0 == True)&(hypothesis1 == True)&(hypotheses == True):\n    # LB: 1.1\n    mean_vote_ratio = {'seizure_vote':  0.196002,\n    'gpd_vote':0.156386,\n    'lrda_vote': 0.155805,\n    'other_vote': 0.17610,\n    'grda_vote': 0.17660,\n    'lpd_vote': 0.139101}\nelif (hypothesis0 == True)&(hypothesis1 == False)&(hypotheses == True):\n    # LB: 1.0\n    mean_vote_ratio = {'seizure_vote':    0.174031,\n    'lpd_vote':        0.112700,\n    'gpd_vote':        0.090854,\n    'lrda_vote':       0.071484,\n    'grda_vote':       0.136408,\n    'other_vote':      0.414523,}\nelif (hypothesis0 == False)&(hypothesis1 == True)&(hypotheses == True):\n    # LB: 0.97\n    mean_vote_ratio = {'seizure_vote':    0.152810,\n    'lpd_vote':        0.142456,\n    'gpd_vote':        0.104062,\n    'lrda_vote':       0.065407,\n    'grda_vote':       0.114851,\n    'other_vote':      0.420414,}\nelif (hypothesis0 == False)&(hypothesis1 == False)&(hypotheses == True):\n    # LB: 1.28\n    mean_vote_ratio = {'seizure_vote': 0.310718,\n    'lpd_vote': 0.046279,\n    'gpd_vote': 0.051885,\n    'lrda_vote': 0.081796,\n    'grda_vote': 0.231471,\n    'other_vote': 0.277851,}\nelse:\n    # sub will fail\n    mean_vote_ratio = {'seizure_vote': np.nan,\n    'lpd_vote': np.nan,\n    'gpd_vote': np.nan,\n    'lrda_vote': np.nan,\n    'grda_vote': np.nan,\n    'other_vote': np.nan,}\n\n\ntargets = ['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\nfor target in targets:\n    sub[target] = mean_vote_ratio[target]\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2024-01-28T07:08:09.664677Z","iopub.execute_input":"2024-01-28T07:08:09.665324Z","iopub.status.idle":"2024-01-28T07:08:09.695896Z","shell.execute_reply.started":"2024-01-28T07:08:09.665274Z","shell.execute_reply":"2024-01-28T07:08:09.694804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv(\"/kaggle/working/submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-01-28T07:08:09.697293Z","iopub.execute_input":"2024-01-28T07:08:09.697634Z","iopub.status.idle":"2024-01-28T07:08:09.704253Z","shell.execute_reply.started":"2024-01-28T07:08:09.697602Z","shell.execute_reply":"2024-01-28T07:08:09.703194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}