{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 0 IMPORT","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom fastai.basics import *\nfrom fastai.callback.all import *\nfrom fastai.vision.all import *\nfrom fastai.medical.imaging import *\nimport pydicom","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:51.509985Z","iopub.execute_input":"2021-07-12T11:39:51.510629Z","iopub.status.idle":"2021-07-12T11:39:53.104111Z","shell.execute_reply.started":"2021-07-12T11:39:51.510493Z","shell.execute_reply":"2021-07-12T11:39:53.102979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1 LOADING THE DATASET","metadata":{}},{"cell_type":"code","source":"train_image_path = \"../input/siim-covid19-detection/train_image_level.csv\"\nsample_sub_path = \"../input/siim-covid19-detection/sample_submission.csv\"\ntrain_study_path = \"../input/siim-covid19-detection/train_study_level.csv\"\nprint (\"__study level csv__\")\ntrain_study_level = pd.read_csv(train_study_path)\nprint (train_study_level.head())\n\nprint (\"__sample submission__\")\nsample_sub = pd.read_csv(sample_sub_path)\nprint (sample_sub.head())\n\nprint (\"__train level csv__\")\ntrain_image_level = pd.read_csv(train_image_path)\nprint (train_image_level.head())","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:41:25.998094Z","iopub.execute_input":"2021-07-12T11:41:25.998713Z","iopub.status.idle":"2021-07-12T11:41:26.070259Z","shell.execute_reply.started":"2021-07-12T11:41:25.99866Z","shell.execute_reply":"2021-07-12T11:41:26.069144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2 WORKING ON DATAFRAMES","metadata":{}},{"cell_type":"code","source":"#rename the column to merge the dataframes\ntrain_study_level.rename(columns = {'id': 'StudyInstanceUID'}, inplace =True)\ntrain_study_level[:1]","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:53.17896Z","iopub.execute_input":"2021-07-12T11:39:53.17972Z","iopub.status.idle":"2021-07-12T11:39:53.195793Z","shell.execute_reply.started":"2021-07-12T11:39:53.179664Z","shell.execute_reply":"2021-07-12T11:39:53.194737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#remove the _study in 'StudyInstanceUID'\ntrain_study_level['StudyInstanceUID'] = train_study_level['StudyInstanceUID'].str.strip('_study')\ntrain_study_level[:5]","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:53.197528Z","iopub.execute_input":"2021-07-12T11:39:53.197866Z","iopub.status.idle":"2021-07-12T11:39:53.217704Z","shell.execute_reply.started":"2021-07-12T11:39:53.197835Z","shell.execute_reply":"2021-07-12T11:39:53.216467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#merging the two dataframes\ntrain_df = train_image_level.merge(train_study_level)\ntrain_df[:1]","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:53.219098Z","iopub.execute_input":"2021-07-12T11:39:53.219403Z","iopub.status.idle":"2021-07-12T11:39:53.244662Z","shell.execute_reply.started":"2021-07-12T11:39:53.219373Z","shell.execute_reply":"2021-07-12T11:39:53.243502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We then clean up the dataframe. We drop a few columns and make a single column as class label. We also drop `boxes` since we can use `label` to obtain coordinates for binding boxes.","metadata":{}},{"cell_type":"code","source":"train_df['id'] = train_df['id'].str.strip('_image')\ntrain_df.loc[train_df['Negative for Pneumonia']==1, 'class_y'] = 'Negative'\ntrain_df.loc[train_df['Typical Appearance']==1, 'class_y'] = 'Typical'\ntrain_df.loc[train_df['Indeterminate Appearance']==1, 'class_y'] = 'Indeterminate'\ntrain_df.loc[train_df['Atypical Appearance']==1, 'class_y'] = 'Atypical'\ntrain_df.drop(['boxes', 'Negative for Pneumonia', 'Typical Appearance', \n             'Indeterminate Appearance', 'Atypical Appearance', 'StudyInstanceUID'], axis=1, inplace=True)\ntrain_df[:5]","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:53.24644Z","iopub.execute_input":"2021-07-12T11:39:53.246909Z","iopub.status.idle":"2021-07-12T11:39:53.279586Z","shell.execute_reply.started":"2021-07-12T11:39:53.246861Z","shell.execute_reply":"2021-07-12T11:39:53.278298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.class_y.value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:53.281025Z","iopub.execute_input":"2021-07-12T11:39:53.281368Z","iopub.status.idle":"2021-07-12T11:39:53.291476Z","shell.execute_reply.started":"2021-07-12T11:39:53.281332Z","shell.execute_reply":"2021-07-12T11:39:53.29066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#identifying number of boxes\nnum_of_boxes = []\nfor i in train_df.index:\n    label_len = len(train_df.label[i].split(' '))\n    num_box = label_len//6\n    num_of_boxes.append(num_box)","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:53.292882Z","iopub.execute_input":"2021-07-12T11:39:53.293408Z","iopub.status.idle":"2021-07-12T11:39:53.39319Z","shell.execute_reply.started":"2021-07-12T11:39:53.293371Z","shell.execute_reply":"2021-07-12T11:39:53.392054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['num_of_boxes'] = num_of_boxes\ntrain_df.head() ","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:53.395844Z","iopub.execute_input":"2021-07-12T11:39:53.396218Z","iopub.status.idle":"2021-07-12T11:39:53.413041Z","shell.execute_reply.started":"2021-07-12T11:39:53.396182Z","shell.execute_reply":"2021-07-12T11:39:53.411865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.num_of_boxes.value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:53.414794Z","iopub.execute_input":"2021-07-12T11:39:53.415216Z","iopub.status.idle":"2021-07-12T11:39:53.428612Z","shell.execute_reply.started":"2021-07-12T11:39:53.415179Z","shell.execute_reply":"2021-07-12T11:39:53.427565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we extract co-ordinates of the binding boxes from `label`","metadata":{}},{"cell_type":"code","source":"\nlabel_COORD = []\nfor i in train_df.index:\n    num_of_boxes = train_df.num_of_boxes[i]\n    val = train_df.label[i].split(' ')\n    if num_of_boxes == 1 : coord = val[2:6]\n    if num_of_boxes == 2 : coord = val[2:6] + val [8:12]\n    if num_of_boxes == 3 : coord = val[2:6] + val [8:12] + val [14:18]\n    if num_of_boxes == 4 : coord = val[2:6] + val [8:12] + val [14:18] + val[20:24]\n    if num_of_boxes == 5 : coord = val[2:6] + val [8:12] + val [14:18] + val[20:24] + val[26:30]\n    label_COORD.append(coord)\n     ","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:53.429743Z","iopub.execute_input":"2021-07-12T11:39:53.430165Z","iopub.status.idle":"2021-07-12T11:39:53.625918Z","shell.execute_reply.started":"2021-07-12T11:39:53.430132Z","shell.execute_reply":"2021-07-12T11:39:53.624838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['label_COORD'] = label_COORD\ndel train_df['label']\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:53.627174Z","iopub.execute_input":"2021-07-12T11:39:53.627506Z","iopub.status.idle":"2021-07-12T11:39:53.649237Z","shell.execute_reply.started":"2021-07-12T11:39:53.627468Z","shell.execute_reply":"2021-07-12T11:39:53.6481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Rename the column `id` to `SOPInstanceUID` to merge the dataframe with meta data","metadata":{}},{"cell_type":"code","source":"train_df.rename(columns = {'id':'SOPInstanceUID'},inplace = True)\ntrain_df[:1]","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:53.651785Z","iopub.execute_input":"2021-07-12T11:39:53.652277Z","iopub.status.idle":"2021-07-12T11:39:53.666444Z","shell.execute_reply.started":"2021-07-12T11:39:53.652232Z","shell.execute_reply":"2021-07-12T11:39:53.665109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3 LOADING THE META DATA","metadata":{}},{"cell_type":"markdown","source":"[We then load the DICOM metadata that we have obtained](https://www.kaggle.com/slimshadymm/visualizing-dicoms) ","metadata":{}},{"cell_type":"code","source":"dicom_df = pd.read_pickle('../input/visualizing-dicoms/dicoms_df.pkl')\ndicom_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:53.668326Z","iopub.execute_input":"2021-07-12T11:39:53.669079Z","iopub.status.idle":"2021-07-12T11:39:55.250068Z","shell.execute_reply.started":"2021-07-12T11:39:53.669029Z","shell.execute_reply":"2021-07-12T11:39:55.248737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_merge = pd.merge(dicom_df,train_df, on = 'SOPInstanceUID')\ndicom_merge[:1]","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:55.251708Z","iopub.execute_input":"2021-07-12T11:39:55.252158Z","iopub.status.idle":"2021-07-12T11:39:55.330971Z","shell.execute_reply.started":"2021-07-12T11:39:55.252108Z","shell.execute_reply":"2021-07-12T11:39:55.329966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Save the dataframe as `.csv` file. Before that, we check that the file path `fname` corresponds to the correct `SOPINstanceUID`","metadata":{}},{"cell_type":"code","source":"dicom_merge['fname'][100]","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:55.332275Z","iopub.execute_input":"2021-07-12T11:39:55.332586Z","iopub.status.idle":"2021-07-12T11:39:55.339181Z","shell.execute_reply.started":"2021-07-12T11:39:55.332556Z","shell.execute_reply":"2021-07-12T11:39:55.338188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_merge['SOPInstanceUID'][100]","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:55.340388Z","iopub.execute_input":"2021-07-12T11:39:55.3407Z","iopub.status.idle":"2021-07-12T11:39:55.352533Z","shell.execute_reply.started":"2021-07-12T11:39:55.340672Z","shell.execute_reply":"2021-07-12T11:39:55.351377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_merge.to_csv('dicom_merge.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2021-07-12T11:39:55.353905Z","iopub.execute_input":"2021-07-12T11:39:55.354197Z","iopub.status.idle":"2021-07-12T11:39:56.686397Z","shell.execute_reply.started":"2021-07-12T11:39:55.35417Z","shell.execute_reply":"2021-07-12T11:39:56.685005Z"},"trusted":true},"execution_count":null,"outputs":[]}]}