{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import modules","metadata":{}},{"cell_type":"code","source":"import os\nimport shutil\nimport glob\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport joblib","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-02T11:26:48.606493Z","iopub.execute_input":"2024-04-02T11:26:48.606940Z","iopub.status.idle":"2024-04-02T11:26:49.230239Z","shell.execute_reply.started":"2024-04-02T11:26:48.606904Z","shell.execute_reply":"2024-04-02T11:26:49.228782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Set global parameter","metadata":{}},{"cell_type":"code","source":"BASE_DIR = Path(\"/kaggle/input/hms-harmful-brain-activity-classification\")\n\nEEG_SAMPLING_TIME = 50  #second\nEEG_SAMPLING_RATE = 200 #Hz\nEEG_DURATION = EEG_SAMPLING_RATE * EEG_SAMPLING_TIME","metadata":{"execution":{"iopub.status.busy":"2024-04-02T11:26:49.233319Z","iopub.execute_input":"2024-04-02T11:26:49.234745Z","iopub.status.idle":"2024-04-02T11:26:49.241905Z","shell.execute_reply.started":"2024-04-02T11:26:49.234651Z","shell.execute_reply":"2024-04-02T11:26:49.240800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(BASE_DIR/\"train.csv\")\nprint(f\"{len(train_df)=}\")\nprint(f\"train df nan:{train_df.isna().sum().sum()}\")\ntrain_df.head(100)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T11:26:49.243392Z","iopub.execute_input":"2024-04-02T11:26:49.244623Z","iopub.status.idle":"2024-04-02T11:26:49.686223Z","shell.execute_reply.started":"2024-04-02T11:26:49.244573Z","shell.execute_reply":"2024-04-02T11:26:49.684815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_unique = train_df.expert_consensus.unique().tolist()\nlabel_unique","metadata":{"execution":{"iopub.status.busy":"2024-04-02T11:27:55.783375Z","iopub.execute_input":"2024-04-02T11:27:55.783831Z","iopub.status.idle":"2024-04-02T11:27:55.800570Z","shell.execute_reply.started":"2024-04-02T11:27:55.783792Z","shell.execute_reply":"2024-04-02T11:27:55.798978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vote_df = train_df[list(map(lambda s: s.lower()+'_vote', label_unique))].copy()\nratiovote_df = vote_df.apply(lambda x: x/x.sum(), axis=1)\nexpert_consensus_prob_series = ratiovote_df.apply(lambda x: x.max(), axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T11:37:55.589002Z","iopub.execute_input":"2024-04-02T11:37:55.589460Z","iopub.status.idle":"2024-04-02T11:38:25.607196Z","shell.execute_reply.started":"2024-04-02T11:37:55.589425Z","shell.execute_reply":"2024-04-02T11:38:25.606041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"expert_consensus_prob_series.plot.hist(bins=10)","metadata":{"execution":{"iopub.status.busy":"2024-04-02T11:40:57.648028Z","iopub.execute_input":"2024-04-02T11:40:57.648418Z","iopub.status.idle":"2024-04-02T11:40:57.976107Z","shell.execute_reply.started":"2024-04-02T11:40:57.648386Z","shell.execute_reply":"2024-04-02T11:40:57.974895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.loc[expert_consensus_prob_series.sort_values(ascending=True).index.tolist()].to_csv('not_consensus.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-02T11:59:19.791461Z","iopub.execute_input":"2024-04-02T11:59:19.792648Z","iopub.status.idle":"2024-04-02T11:59:20.900422Z","shell.execute_reply.started":"2024-04-02T11:59:19.792599Z","shell.execute_reply":"2024-04-02T11:59:20.899090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_train_eeg(q:dict) -> pd.DataFrame:\n    parquet_df = pd.read_parquet(BASE_DIR/f\"train_eegs/{q['eeg_id']}.parquet\")\n    eeg_start_index = int(EEG_SAMPLING_RATE * q[\"eeg_label_offset_seconds\"])\n    return parquet_df.iloc[eeg_start_index:eeg_start_index+EEG_DURATION]\n\ndef plot_eeg(df, moving_avg=1):\n    fig, axs = plt.subplots(20, 1, figsize=(30, 15), sharex=True)\n    for i, ax in enumerate(axs):\n        ax.plot(df.iloc[:,i], color=\"black\")\n        for vline in df[df.iloc[:,i].isna()].index:\n            line_min = df.iloc[:,i].min()\n            line_max = df.iloc[:,i].max()\n            ax.vlines(vline, line_min, line_max, color='red')\n        ax.set_ylabel(df.columns[i], rotation=0)\n        ax.set_yticklabels([])\n        ax.set_yticks([])\n        ax.set_xticks([])\n        ax.spines[[\"top\", \"bottom\", \"left\", \"right\"]].set_visible(False)","metadata":{"execution":{"iopub.status.busy":"2024-01-22T14:22:23.814970Z","iopub.execute_input":"2024-01-22T14:22:23.815339Z","iopub.status.idle":"2024-01-22T14:22:23.826134Z","shell.execute_reply.started":"2024-01-22T14:22:23.815309Z","shell.execute_reply":"2024-01-22T14:22:23.824841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_all_eeg_img(p_eeg):\n    eeg = pd.read_parquet(p_eeg)\n    plot_eeg(eeg)\n    plt.savefig(f'all_eeg_figs/{p_eeg.split(\"/\")[-1].split(\".\")[0]}.png')\n    plt.close()","metadata":{"execution":{"iopub.status.busy":"2024-01-22T14:22:23.827526Z","iopub.execute_input":"2024-01-22T14:22:23.827858Z","iopub.status.idle":"2024-01-22T14:22:23.835209Z","shell.execute_reply.started":"2024-01-22T14:22:23.827830Z","shell.execute_reply":"2024-01-22T14:22:23.834217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pywt\n\ndef maddest(d, axis=None):\n    return np.mean(np.absolute(d - np.mean(d, axis)), axis)\n\ndef denoise(x, wavelet='db8', level=1): # dmeyがてんかん患者のEEG信号からのノイズ除去に最適であり、db8が健康な被験者のEEG信号からのノイズ除去に最適\n    ret = {key:[] for key in x.columns}\n    for pos in x.columns:\n        coeff = pywt.wavedec(x[pos], wavelet, mode=\"per\")\n        sigma = (1/0.6745) * maddest(coeff[-level])\n        uthresh = sigma * np.sqrt(2*np.log(len(x)))\n        coeff[1:] = (pywt.threshold(i, value=uthresh, mode='hard') for i in coeff[1:])\n        ret[pos]=pywt.waverec(coeff, wavelet, mode='per')\n    return pd.DataFrame(ret)","metadata":{"execution":{"iopub.status.busy":"2024-01-22T14:22:23.836360Z","iopub.execute_input":"2024-01-22T14:22:23.836943Z","iopub.status.idle":"2024-01-22T14:22:24.332725Z","shell.execute_reply.started":"2024-01-22T14:22:23.836914Z","shell.execute_reply":"2024-01-22T14:22:24.331712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"isna# Sample Vizulization","metadata":{}},{"cell_type":"code","source":"q = train_df.iloc[10]\nq = dict(q)\nq[\"expert_consensus\"]","metadata":{"execution":{"iopub.status.busy":"2024-01-22T14:22:24.334571Z","iopub.execute_input":"2024-01-22T14:22:24.334918Z","iopub.status.idle":"2024-01-22T14:22:24.341995Z","shell.execute_reply.started":"2024-01-22T14:22:24.334887Z","shell.execute_reply":"2024-01-22T14:22:24.340949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## raw data","metadata":{}},{"cell_type":"code","source":"eeg = get_train_eeg(q)\nplot_eeg(eeg)","metadata":{"execution":{"iopub.status.busy":"2024-01-22T14:22:24.345024Z","iopub.execute_input":"2024-01-22T14:22:24.345417Z","iopub.status.idle":"2024-01-22T14:22:26.302168Z","shell.execute_reply.started":"2024-01-22T14:22:24.345374Z","shell.execute_reply":"2024-01-22T14:22:26.300964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## db8 is best for removing noise from EEG signals of healthy patients","metadata":{}},{"cell_type":"code","source":"db8_processed_eeg = denoise(eeg, wavelet=\"db8\")\nplot_eeg(db8_processed_eeg)","metadata":{"execution":{"iopub.status.busy":"2024-01-22T14:22:26.303408Z","iopub.execute_input":"2024-01-22T14:22:26.303790Z","iopub.status.idle":"2024-01-22T14:22:27.936212Z","shell.execute_reply.started":"2024-01-22T14:22:26.303758Z","shell.execute_reply":"2024-01-22T14:22:27.935277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## dmey is best for removing noise from EEG signals of epilepsy patients","metadata":{}},{"cell_type":"code","source":"dmey_processed_eeg = denoise(eeg, wavelet=\"dmey\")\nplot_eeg(dmey_processed_eeg)","metadata":{"execution":{"iopub.status.busy":"2024-01-22T14:22:27.937452Z","iopub.execute_input":"2024-01-22T14:22:27.937982Z","iopub.status.idle":"2024-01-22T14:22:29.853116Z","shell.execute_reply.started":"2024-01-22T14:22:27.937951Z","shell.execute_reply":"2024-01-22T14:22:29.852051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Serch missing value eeg index","metadata":{}},{"cell_type":"code","source":"def detect_na(row):\n    if get_train_eeg(row).isna().sum().sum():\n        return f\"{row['eeg_id']}-{row['eeg_sub_id']}\"","metadata":{"execution":{"iopub.status.busy":"2024-01-22T14:22:29.854443Z","iopub.execute_input":"2024-01-22T14:22:29.855356Z","iopub.status.idle":"2024-01-22T14:22:29.861772Z","shell.execute_reply.started":"2024-01-22T14:22:29.855315Z","shell.execute_reply":"2024-01-22T14:22:29.860719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig_out_dir = \"all_eeg_figs\"\nos.makedirs(fig_out_dir, exist_ok=True)\n\n# I don't like run so long, that's why I make branch of runtype, if you want to run all in interactive, you should use only Batch block.\nif os.environ.get('KAGGLE_KERNEL_RUN_TYPE','') == 'Interactive':\n    print(\"Running on Interactive Notebook\")\n    temp_train_df = train_df.sample(100).copy()\n    drop_index = joblib.Parallel(n_jobs=-1)(joblib.delayed(detect_na)(row) for i, row in temp_train_df.iterrows())\n    joblib.Parallel(n_jobs=-1)(joblib.delayed(save_all_eeg_img)(p_eeg) for p_eeg in glob.glob(os.path.join(BASE_DIR, \"train_eegs/*\"))[:10])\nelif os.environ.get('KAGGLE_KERNEL_RUN_TYPE','') == 'Batch':\n    print(\"Running on Background Notebook\")\n    joblib.Parallel(n_jobs=-1)(joblib.delayed(save_all_eeg_img)(p_eeg) for p_eeg in glob.glob(os.path.join(BASE_DIR, \"train_eegs/*\")))\n    drop_index = joblib.Parallel(n_jobs=-1)(joblib.delayed(detect_na)(row) for i, row in train_df.iterrows())\n    \ndrop_index = list(filter(None, drop_index))","metadata":{"execution":{"iopub.status.busy":"2024-01-22T14:22:29.863090Z","iopub.execute_input":"2024-01-22T14:22:29.864099Z","iopub.status.idle":"2024-01-22T14:22:44.400148Z","shell.execute_reply.started":"2024-01-22T14:22:29.864059Z","shell.execute_reply":"2024-01-22T14:22:44.398798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"detect {len(drop_index)=}\")\njoblib.dump(drop_index, \"dropped-eeg-id-sub.joblib\")","metadata":{"execution":{"iopub.status.busy":"2024-01-22T14:22:44.402204Z","iopub.execute_input":"2024-01-22T14:22:44.402569Z","iopub.status.idle":"2024-01-22T14:22:44.412556Z","shell.execute_reply.started":"2024-01-22T14:22:44.402535Z","shell.execute_reply":"2024-01-22T14:22:44.411232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_missingvalue_set = set()\nfor drp_idx in drop_index:\n    eedid, eegsubid = map(int, drp_idx.split(\"-\"))\n    row = dict(train_df.loc[(train_df.eeg_id==eedid)&(train_df.eeg_sub_id==eegsubid),:].iloc[0])\n    na_eeg = get_train_eeg(row)\n    unique_missingvalue_set.add(*set(na_eeg.isna().sum()))\n    if len(set(na_eeg.isna().sum()))!=1:\n        print(\"partial missing value is exit!\")\nelse:\n    print(\"all missing value is same time line\")\nprint(\"-\"*50)\nprint(f\"each eeg missing value range: {min(unique_missingvalue_set)} ~ {max(unique_missingvalue_set)}\")","metadata":{"execution":{"iopub.status.busy":"2024-01-22T14:22:44.414382Z","iopub.execute_input":"2024-01-22T14:22:44.415509Z","iopub.status.idle":"2024-01-22T14:22:44.505365Z","shell.execute_reply.started":"2024-01-22T14:22:44.415454Z","shell.execute_reply":"2024-01-22T14:22:44.504369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(row[\"expert_consensus\"])\nna_eeg = get_train_eeg(row)\nplot_eeg(na_eeg)","metadata":{"execution":{"iopub.status.busy":"2024-01-22T14:22:44.506782Z","iopub.execute_input":"2024-01-22T14:22:44.507373Z","iopub.status.idle":"2024-01-22T14:22:46.611362Z","shell.execute_reply.started":"2024-01-22T14:22:44.507341Z","shell.execute_reply":"2024-01-22T14:22:46.610308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Search outlier value","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}