{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os, gc\nos.environ[\"CUDA_VISIBLE_DEVICES\"]=\"0,1\"\nimport tensorflow as tf\nimport pandas as pd, numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-03-26T01:33:32.390200Z","iopub.execute_input":"2024-03-26T01:33:32.391307Z","iopub.status.idle":"2024-03-26T01:33:33.773404Z","shell.execute_reply.started":"2024-03-26T01:33:32.391264Z","shell.execute_reply":"2024-03-26T01:33:33.772368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"https://www.kaggle.com/competitions/hms-harmful-brain-activity-classification/discussion/477461\n\nThe discussion above emphasizes the importance of hard samples. Therefore, I examined the distributions of hard samples and soft samples. Thank you","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/train.csv')\nTARGETS = df.columns[-6:]\nprint('Train shape:', df.shape )\nprint('Targets', list(TARGETS))\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-26T01:30:19.940039Z","iopub.execute_input":"2024-03-26T01:30:19.940843Z","iopub.status.idle":"2024-03-26T01:30:20.276206Z","shell.execute_reply.started":"2024-03-26T01:30:19.940803Z","shell.execute_reply":"2024-03-26T01:30:20.274560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_id':'first','spectrogram_label_offset_seconds':'min'})\ntrain.columns = ['spec_id','min']\n\ntmp = df.groupby('eeg_id')[['spectrogram_id','spectrogram_label_offset_seconds']].agg(\n    {'spectrogram_label_offset_seconds':'max'})\ntrain['max'] = tmp\n\ntmp = df.groupby('eeg_id')[['patient_id']].agg('first')\ntrain['patient_id'] = tmp\n\ntmp = df.groupby('eeg_id')[TARGETS].agg('sum')\nfor t in TARGETS:\n    train[t] = tmp[t].values\n    \ny_data = train[TARGETS].values\ny_data = y_data / y_data.sum(axis=1,keepdims=True)\ntrain[TARGETS] = y_data\n\ntmp = df.groupby('eeg_id')[['expert_consensus']].agg('first')\ntrain['target'] = tmp\n\ntrain = train.reset_index()\nprint('Train non-overlapp eeg_id shape:', train.shape )\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-26T01:30:20.277970Z","iopub.execute_input":"2024-03-26T01:30:20.278459Z","iopub.status.idle":"2024-03-26T01:30:20.416459Z","shell.execute_reply.started":"2024-03-26T01:30:20.278409Z","shell.execute_reply":"2024-03-26T01:30:20.415412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_trainset = train.iloc[:,5:]\nsub_trainset.head()\n\nvote_cols = ['seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote',]\nind = 1\nplt.figure(figsize = [20,20])\nfor x_val in vote_cols:\n    for y_val in vote_cols:\n        plt.subplot(6,6,ind)\n        ind += 1\n        plt.title(f'{x_val} vs {y_val}')\n        x,y = sub_trainset[x_val], sub_trainset[y_val]\n        plt.scatter(x,y, s=32, alpha=.8)\n        plt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"execution":{"iopub.status.busy":"2024-03-26T01:30:48.780562Z","iopub.execute_input":"2024-03-26T01:30:48.781232Z","iopub.status.idle":"2024-03-26T01:31:00.355340Z","shell.execute_reply.started":"2024-03-26T01:30:48.781198Z","shell.execute_reply":"2024-03-26T01:31:00.354348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### all samples distribution","metadata":{}},{"cell_type":"code","source":"vote_cols = ['seizure_vote','lpd_vote','gpd_vote','lrda_vote','grda_vote','other_vote',]\nind = 1\nplt.figure(figsize = [20,4])\nfor col_name in vote_cols:\n    plt.subplot(1,6,ind)\n    ind += 1\n    plt.title(f'{col_name}')\n    plt.hist(sub_trainset[col_name], bins = 100)","metadata":{"execution":{"iopub.status.busy":"2024-03-26T01:32:37.287563Z","iopub.execute_input":"2024-03-26T01:32:37.288033Z","iopub.status.idle":"2024-03-26T01:32:40.209463Z","shell.execute_reply.started":"2024-03-26T01:32:37.288002Z","shell.execute_reply":"2024-03-26T01:32:40.208170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### hardsample distribution","metadata":{}},{"cell_type":"code","source":"ind = 1\nplt.figure(figsize = [20,4])\n\nconditional_ind = (sub_trainset.other_vote.values < 0.2) | (sub_trainset.other_vote.values > 0.8)\n\nfor col_name in vote_cols:\n    plt.subplot(1,6,ind)\n    ind += 1\n    plt.title(f'{col_name}')\n    plt.hist(sub_trainset.loc[conditional_ind,col_name], bins = 100)","metadata":{"execution":{"iopub.status.busy":"2024-03-26T01:32:40.212345Z","iopub.execute_input":"2024-03-26T01:32:40.214408Z","iopub.status.idle":"2024-03-26T01:32:42.735197Z","shell.execute_reply.started":"2024-03-26T01:32:40.214356Z","shell.execute_reply":"2024-03-26T01:32:42.733988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### softsample distribution\n* In the soft sample condition, a shift in the distribution of other target labels occurred.\n* In cases of imbalanced aspect labels, the model's training performance decreases.","metadata":{}},{"cell_type":"code","source":"ind = 1\nplt.figure(figsize = [20,4])\n\nconditional_ind = (sub_trainset.other_vote.values > 0.2) & (sub_trainset.other_vote.values < 0.8)\nfor col_name in vote_cols:\n    plt.subplot(1,6,ind)\n    ind += 1\n    plt.title(f'{col_name}')\n    plt.hist(sub_trainset.loc[conditional_ind,col_name], bins = 100)","metadata":{"execution":{"iopub.status.busy":"2024-03-26T01:32:42.737279Z","iopub.execute_input":"2024-03-26T01:32:42.738104Z","iopub.status.idle":"2024-03-26T01:32:45.218552Z","shell.execute_reply.started":"2024-03-26T01:32:42.738048Z","shell.execute_reply":"2024-03-26T01:32:45.216917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The distribution of other votes is a key point","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize = [20,20])\nplt.subplot(3,2,1)\nsns.boxplot(x='target', y='seizure_vote', data=train)\n\nplt.subplot(3,2,2)\nsns.boxplot(x='target', y='lpd_vote', data=train)\n\nplt.subplot(3,2,3)\nsns.boxplot(x='target', y='gpd_vote', data=train)\n\nplt.subplot(3,2,4)\nsns.boxplot(x='target', y='other_vote', data=train)\n\nplt.subplot(3,2,5)\nsns.boxplot(x='target', y='grda_vote', data=train)\n\nplt.subplot(3,2,6)\nsns.boxplot(x='target', y='lrda_vote', data=train)","metadata":{"execution":{"iopub.status.busy":"2024-03-26T01:33:45.882823Z","iopub.execute_input":"2024-03-26T01:33:45.884090Z","iopub.status.idle":"2024-03-26T01:33:47.795153Z","shell.execute_reply.started":"2024-03-26T01:33:45.884043Z","shell.execute_reply":"2024-03-26T01:33:47.793650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}