{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7699086,"sourceType":"datasetVersion","datasetId":4493984}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Purpose\n\nI would like to create ML based on **confident diagnosis!**. The output of this notebook will be used to create ML model.\n\nThis notebook uses result of other notebook [HMS-EDA-SpectrogramLength](https://www.kaggle.com/code/hidetaketakahashi/hms-eda-spectrogramlength). ","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport time\nimport math","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-26T10:02:26.820956Z","iopub.execute_input":"2024-02-26T10:02:26.822514Z","iopub.status.idle":"2024-02-26T10:02:28.028758Z","shell.execute_reply.started":"2024-02-26T10:02:26.822429Z","shell.execute_reply":"2024-02-26T10:02:28.027827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_all = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\ntest_df  = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/test.csv\")\nsample_df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/sample_submission.csv\")\n\ntrain_spectro_dir = \"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/\"\ntrain_spectro_files = os.listdir(train_spectro_dir)\nlen(train_spectro_files)\n","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:28.030488Z","iopub.execute_input":"2024-02-26T10:02:28.031166Z","iopub.status.idle":"2024-02-26T10:02:28.793322Z","shell.execute_reply.started":"2024-02-26T10:02:28.031135Z","shell.execute_reply":"2024-02-26T10:02:28.791916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Import output of other notebook","metadata":{}},{"cell_type":"code","source":"spect_df = pd.read_csv(\"/kaggle/input/hms-processeddata/spectrogram_length.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:28.795936Z","iopub.execute_input":"2024-02-26T10:02:28.796812Z","iopub.status.idle":"2024-02-26T10:02:28.816741Z","shell.execute_reply.started":"2024-02-26T10:02:28.796759Z","shell.execute_reply":"2024-02-26T10:02:28.815482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spect_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:28.821302Z","iopub.execute_input":"2024-02-26T10:02:28.822135Z","iopub.status.idle":"2024-02-26T10:02:28.842676Z","shell.execute_reply.started":"2024-02-26T10:02:28.822085Z","shell.execute_reply":"2024-02-26T10:02:28.841418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Merge it to train data ","metadata":{}},{"cell_type":"code","source":"train_df_all = train_df_all.merge(spect_df, on = \"spectrogram_id\", how = \"left\")","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:28.844966Z","iopub.execute_input":"2024-02-26T10:02:28.845929Z","iopub.status.idle":"2024-02-26T10:02:28.890717Z","shell.execute_reply.started":"2024-02-26T10:02:28.845882Z","shell.execute_reply":"2024-02-26T10:02:28.889346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_all.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:28.892584Z","iopub.execute_input":"2024-02-26T10:02:28.893101Z","iopub.status.idle":"2024-02-26T10:02:28.916236Z","shell.execute_reply.started":"2024-02-26T10:02:28.893054Z","shell.execute_reply":"2024-02-26T10:02:28.914728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Number of symptoms in a spectrogram","metadata":{}},{"cell_type":"code","source":"select = [\"spectrogram_id\",\"expert_consensus\" ]\nspect_numberOfSymptoms = train_df_all[select + [\"length\"]].groupby(select).count().reset_index()[\"spectrogram_id\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:28.918387Z","iopub.execute_input":"2024-02-26T10:02:28.919012Z","iopub.status.idle":"2024-02-26T10:02:28.959955Z","shell.execute_reply.started":"2024-02-26T10:02:28.918974Z","shell.execute_reply":"2024-02-26T10:02:28.958602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nfilter1 = spect_numberOfSymptoms == 1\nspect_numberOfSymptoms.loc[~filter1].hist(ax = ax)\nspect_numberOfSymptoms.loc[filter1].hist(ax = ax, bins = 3)\nax.set_title(\"Most of spectrograms has one symptom\")\nax.set_xlabel(\"Number of symptoms in a spectrogram\")\nax.set_ylabel(\"count of spectrograms\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:28.961697Z","iopub.execute_input":"2024-02-26T10:02:28.962059Z","iopub.status.idle":"2024-02-26T10:02:29.412081Z","shell.execute_reply.started":"2024-02-26T10:02:28.962030Z","shell.execute_reply":"2024-02-26T10:02:29.410389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To extract Spectrogram IDs with a single symptom","metadata":{}},{"cell_type":"code","source":"filter1 = spect_numberOfSymptoms == 1\nspect_singleSymptom = set(spect_numberOfSymptoms.loc[filter1].index)","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:29.413922Z","iopub.execute_input":"2024-02-26T10:02:29.414591Z","iopub.status.idle":"2024-02-26T10:02:29.424909Z","shell.execute_reply.started":"2024-02-26T10:02:29.414549Z","shell.execute_reply":"2024-02-26T10:02:29.423586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(spect_singleSymptom)","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:29.429522Z","iopub.execute_input":"2024-02-26T10:02:29.429977Z","iopub.status.idle":"2024-02-26T10:02:29.439275Z","shell.execute_reply.started":"2024-02-26T10:02:29.429942Z","shell.execute_reply":"2024-02-26T10:02:29.437796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Confidence of vote","metadata":{}},{"cell_type":"code","source":"train_df_all[\"expert_consensus\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:29.440820Z","iopub.execute_input":"2024-02-26T10:02:29.441224Z","iopub.status.idle":"2024-02-26T10:02:29.475618Z","shell.execute_reply.started":"2024-02-26T10:02:29.441189Z","shell.execute_reply":"2024-02-26T10:02:29.474212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vote_cnames = ['seizure_vote', 'lpd_vote',\n       'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']\n\ntotal_vote = np.sum(train_df_all[vote_cnames], axis =1).to_numpy()\nvote_mat = train_df_all[vote_cnames].to_numpy().astype(float)\nvote_confidence = vote_mat/(total_vote.reshape(-1,1))\nmax_vote_rate = np.max(vote_confidence, axis = 1)","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:29.477261Z","iopub.execute_input":"2024-02-26T10:02:29.477633Z","iopub.status.idle":"2024-02-26T10:02:29.528625Z","shell.execute_reply.started":"2024-02-26T10:02:29.477604Z","shell.execute_reply":"2024-02-26T10:02:29.527300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N = len(vote_cnames)\nncol = 3\nnrow = math.ceil(N/ncol)\nfig, ax = plt.subplots(nrow, ncol,  figsize = (12, nrow*3.3))\nfor k in range(N):\n    i =  int(k/ncol)\n    j =  k % ncol\n    filter1 = (vote_confidence[:,k] > 0) & (total_vote >= 3)\n    ax[i, j].hist(vote_confidence[:,k][filter1], density = True, bins = 10)\n    ax[i, j].set_title(vote_cnames[k] + \" rate\")\n    ax[i,j].set_xlim(0, 1)","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:29.529990Z","iopub.execute_input":"2024-02-26T10:02:29.530322Z","iopub.status.idle":"2024-02-26T10:02:30.918292Z","shell.execute_reply.started":"2024-02-26T10:02:29.530294Z","shell.execute_reply":"2024-02-26T10:02:30.917172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filter1 = max_vote_rate >= 1.\nfilter2 = total_vote >= 3\nfilter_confident = (filter1 & filter2)\n\ntrain_df_all.loc[filter_confident][\"expert_consensus\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:30.919946Z","iopub.execute_input":"2024-02-26T10:02:30.921009Z","iopub.status.idle":"2024-02-26T10:02:30.950621Z","shell.execute_reply.started":"2024-02-26T10:02:30.920956Z","shell.execute_reply":"2024-02-26T10:02:30.949078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Writing Result of EDA","metadata":{}},{"cell_type":"code","source":"train_df_all[\"total_vote\"] = total_vote\ntrain_df_all[\"max_vote_rate\"] = max_vote_rate","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:30.952528Z","iopub.execute_input":"2024-02-26T10:02:30.952908Z","iopub.status.idle":"2024-02-26T10:02:30.960691Z","shell.execute_reply.started":"2024-02-26T10:02:30.952879Z","shell.execute_reply":"2024-02-26T10:02:30.959108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filter1 = train_df_all[\"spectrogram_id\"].apply(lambda x: x in spect_singleSymptom)\ntrain_df_all[\"SpectrogramWithSingleSymtom\"] = filter1","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:30.962649Z","iopub.execute_input":"2024-02-26T10:02:30.963063Z","iopub.status.idle":"2024-02-26T10:02:31.037258Z","shell.execute_reply.started":"2024-02-26T10:02:30.963029Z","shell.execute_reply":"2024-02-26T10:02:31.035954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_all.to_csv(\"hms_train_eda.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:31.039038Z","iopub.execute_input":"2024-02-26T10:02:31.039414Z","iopub.status.idle":"2024-02-26T10:02:32.284459Z","shell.execute_reply.started":"2024-02-26T10:02:31.039370Z","shell.execute_reply":"2024-02-26T10:02:32.283161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_all.head()","metadata":{"execution":{"iopub.status.busy":"2024-02-26T10:02:32.286038Z","iopub.execute_input":"2024-02-26T10:02:32.286453Z","iopub.status.idle":"2024-02-26T10:02:32.311518Z","shell.execute_reply.started":"2024-02-26T10:02:32.286370Z","shell.execute_reply":"2024-02-26T10:02:32.309705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}