{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Data Analysis**","metadata":{}},{"cell_type":"markdown","source":"* [Dependencies and imports](#section-one)\n* [Read Data](#section-two)\n    * [Study-level](#section-two-one)\n    * [Image-level](#section-two-two)\n    * [Merge study and image levels](#section-two-three)\n* [Data Analysis](#section-three)\n    * [Null values](#section-three-one)\n    * [Duplicate values](#section-three-two)\n    * [Number of images per study](#section-three-three)\n    * [Number of bboxes per image](#section-three-four)\n    * [Study-level class frequency](#section-three-five)\n    * [Image-level class frequency](#section-three-six)\n    * [Class distribution with no bbox](#section-three-seven)\n    * [Number of boxes for the different classes](#section-three-eight)\n    * [Relation between box size and number of boxes per image](#section-three-nine)\n* [DICOM files](#section-four)\n    * [View dicom files metadata](#section-four-one)\n    * [Add metadata and image shape to train df](#section-four-two)\n* [Explore images](#section-five)\n    * [Number of images per study](#section-five-one)\n    * [Studies with 3 images](#section-five-two)\n    * [Studies with 4 images](#section-five-three)\n    * [Studies with 5 images](#section-five-four)\n    * [Studies with 6 images](#section-five-five)\n    * [Studies with 7 images](#section-five-six)\n    * [Studies with 9 images](#section-five-seven)\n    * [Studies with 2 images](#section-five-eight)\n    * [Box count per class](#section-five-nine)\n* [Final train df](#section-six)\n* [Create test df](#section-seven)","metadata":{}},{"cell_type":"markdown","source":"<a id=\"section-one\"></a>\n## **Dependencies and imports**","metadata":{}},{"cell_type":"code","source":"conda install gdcm -c conda-forge","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:04:18.671546Z","iopub.execute_input":"2021-09-22T12:04:18.672056Z","iopub.status.idle":"2021-09-22T12:05:27.129667Z","shell.execute_reply.started":"2021-09-22T12:04:18.671928Z","shell.execute_reply":"2021-09-22T12:05:27.128477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --upgrade --force-reinstall numpy","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:27.131454Z","iopub.execute_input":"2021-09-22T12:05:27.131732Z","iopub.status.idle":"2021-09-22T12:05:41.090960Z","shell.execute_reply.started":"2021-09-22T12:05:27.131698Z","shell.execute_reply":"2021-09-22T12:05:41.089798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport pydicom\nimport numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport cv2\nimport ast\nimport os\nfrom termcolor import colored\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:41.093577Z","iopub.execute_input":"2021-09-22T12:05:41.093868Z","iopub.status.idle":"2021-09-22T12:05:42.379723Z","shell.execute_reply.started":"2021-09-22T12:05:41.093836Z","shell.execute_reply":"2021-09-22T12:05:42.378970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-two\"></a>\n## **Read Data**","metadata":{}},{"cell_type":"code","source":"data_path = '../input/siim-covid19-detection'\noutput_path = './'","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:42.381392Z","iopub.execute_input":"2021-09-22T12:05:42.381967Z","iopub.status.idle":"2021-09-22T12:05:42.387290Z","shell.execute_reply.started":"2021-09-22T12:05:42.381921Z","shell.execute_reply":"2021-09-22T12:05:42.386030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(data_path)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-09-22T12:05:42.388742Z","iopub.execute_input":"2021-09-22T12:05:42.389480Z","iopub.status.idle":"2021-09-22T12:05:42.403788Z","shell.execute_reply.started":"2021-09-22T12:05:42.389435Z","shell.execute_reply":"2021-09-22T12:05:42.402674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = os.path.join(data_path, 'train')\ntrain_study_path = os.path.join(data_path, 'train_study_level.csv')\ntrain_image_path = os.path.join(data_path, 'train_image_level.csv')\n\ntrain_study_df = pd.read_csv(train_study_path)\ntrain_image_df = pd.read_csv(train_image_path)\n\nprint(\"Train Study Shape: {}\\nTrain Image Shape:{}\".format(train_study_df.shape, train_image_df.shape))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:42.405242Z","iopub.execute_input":"2021-09-22T12:05:42.405929Z","iopub.status.idle":"2021-09-22T12:05:42.473541Z","shell.execute_reply.started":"2021-09-22T12:05:42.405862Z","shell.execute_reply":"2021-09-22T12:05:42.472476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have more studies than images","metadata":{}},{"cell_type":"markdown","source":"<a id=\"section-two-one\"></a>\n#### **Study-level**","metadata":{}},{"cell_type":"code","source":"print(\"Train study:\\n\")\ntrain_study_df","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:42.475024Z","iopub.execute_input":"2021-09-22T12:05:42.475630Z","iopub.status.idle":"2021-09-22T12:05:42.503432Z","shell.execute_reply.started":"2021-09-22T12:05:42.475587Z","shell.execute_reply":"2021-09-22T12:05:42.502143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"rename studies ids (get only study id)","metadata":{}},{"cell_type":"code","source":"# rename id col\ntrain_study_df = train_study_df.rename(columns = {'id': 'study_id'}, inplace = False)\n# get only the id (split by '_' to id and 'study' and get the first)\ntrain_study_df[\"study_id\"] = train_study_df[\"study_id\"].apply(lambda x: x.split(\"_\")[0])","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:42.507035Z","iopub.execute_input":"2021-09-22T12:05:42.507358Z","iopub.status.idle":"2021-09-22T12:05:42.518703Z","shell.execute_reply.started":"2021-09-22T12:05:42.507329Z","shell.execute_reply":"2021-09-22T12:05:42.517880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"rename class columns and add int label and class label column","metadata":{}},{"cell_type":"code","source":"NEGATIVE = 'negative'\nTYPICAL = 'typical'\nINDERTEMINATE = 'indeterminate'\nATYPICAL = 'atypical'\n\nstudy_level_labels = {NEGATIVE:0, TYPICAL:1, INDERTEMINATE:2, ATYPICAL:3}\n\n# rename columns for easier use\ntrain_study_df = train_study_df.rename(columns = {'Negative for Pneumonia': NEGATIVE,\n                                                  'Typical Appearance': TYPICAL,\n                                                  'Indeterminate Appearance': INDERTEMINATE,\n                                                  'Atypical Appearance': ATYPICAL}, inplace = False)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:42.520960Z","iopub.execute_input":"2021-09-22T12:05:42.521777Z","iopub.status.idle":"2021-09-22T12:05:42.532000Z","shell.execute_reply.started":"2021-09-22T12:05:42.521729Z","shell.execute_reply":"2021-09-22T12:05:42.531202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = []\nint_labels = []\n\nfor index, row in train_study_df.iterrows():\n    if row[NEGATIVE] == 1:\n        labels.append(NEGATIVE)\n        int_labels.append(study_level_labels[NEGATIVE])\n    elif row[TYPICAL] == 1:\n        labels.append(TYPICAL)\n        int_labels.append(study_level_labels[TYPICAL])\n    elif row[INDERTEMINATE] == 1:\n        labels.append(INDERTEMINATE)\n        int_labels.append(study_level_labels[INDERTEMINATE])\n    elif row[ATYPICAL] == 1:\n        labels.append(ATYPICAL)\n        int_labels.append(study_level_labels[ATYPICAL])\n\ntrain_study_df['study_level'] =  labels\ntrain_study_df['int_label'] =  int_labels","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:42.533443Z","iopub.execute_input":"2021-09-22T12:05:42.534054Z","iopub.status.idle":"2021-09-22T12:05:43.105603Z","shell.execute_reply.started":"2021-09-22T12:05:42.534011Z","shell.execute_reply":"2021-09-22T12:05:43.104547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:43.106855Z","iopub.execute_input":"2021-09-22T12:05:43.107169Z","iopub.status.idle":"2021-09-22T12:05:43.119703Z","shell.execute_reply.started":"2021-09-22T12:05:43.107126Z","shell.execute_reply":"2021-09-22T12:05:43.118646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-two-two\"></a>\n#### **Image-level**","metadata":{}},{"cell_type":"code","source":"print(\"Train image:\\n\") \ntrain_image_df","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:43.121320Z","iopub.execute_input":"2021-09-22T12:05:43.121668Z","iopub.status.idle":"2021-09-22T12:05:43.143373Z","shell.execute_reply.started":"2021-09-22T12:05:43.121635Z","shell.execute_reply":"2021-09-22T12:05:43.142313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"rename images ids (get only image id)","metadata":{}},{"cell_type":"code","source":"# rename id col\ntrain_image_df = train_image_df.rename(columns = {'id': 'img_id'}, inplace = False)\n# get only the id (split by '_' to id and 'image' and get the first)\ntrain_image_df[\"img_id\"] = train_image_df[\"img_id\"].apply(lambda x: x.split(\"_\")[0])","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:43.144780Z","iopub.execute_input":"2021-09-22T12:05:43.145145Z","iopub.status.idle":"2021-09-22T12:05:43.159881Z","shell.execute_reply.started":"2021-09-22T12:05:43.145112Z","shell.execute_reply":"2021-09-22T12:05:43.159027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"rename class columns and add split label to class, score and bboxes columns","metadata":{}},{"cell_type":"code","source":"NONE = 'none'\nOPACITY = 'opacity'\n\nIMAGE_LEVEL_LABEL_SIZE = 4","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:43.161595Z","iopub.execute_input":"2021-09-22T12:05:43.162086Z","iopub.status.idle":"2021-09-22T12:05:43.175677Z","shell.execute_reply.started":"2021-09-22T12:05:43.162034Z","shell.execute_reply":"2021-09-22T12:05:43.174817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_num_boxes(sample):\n    if(isinstance(sample['boxes'], str)): # not nan\n        bboxes = ast.literal_eval(sample['boxes'])\n        return len(bboxes)\n    return 0 # no boxes\n\ndef get_coco_format(sample):\n    if(isinstance(sample['boxes'], str)): # not nan\n        boxes = ast.literal_eval(sample['boxes'])\n        coco_boxes = []\n        for box in boxes:\n            coco_boxes.append([float(box['x']), float(box['y']), float(box['width']), float(box['height'])])\n        return coco_boxes\n    return np.nan\n\ndef get_label(sample, num_boxes):\n    num_components = 6 # opacity/none, score, x1, y1, x2, y2\n    if num_boxes==0:\n        num_boxes = 1 # for no boxes we label [0,0,1,1]\n    label_data = sample['label'].split(' ')\n    label = label_data[0]\n    confidence_scores = []\n    pascal_voc_boxes = []\n    \n    for i in range(num_boxes):\n        start = i*num_components + 1\n        confidence_scores.append(float(label_data[start]))\n        pascal_voc_boxes.append([float(label_data[start+1]), float(label_data[start+2]), float(label_data[start+3]), float(label_data[start+4])])\n    return label, confidence_scores, pascal_voc_boxes","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:43.176768Z","iopub.execute_input":"2021-09-22T12:05:43.177196Z","iopub.status.idle":"2021-09-22T12:05:43.188536Z","shell.execute_reply.started":"2021-09-22T12:05:43.177141Z","shell.execute_reply":"2021-09-22T12:05:43.187358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_num_boxes = []\nall_labels = []\nall_scores = []\nall_pascal_voc_boxes = []\nall_coco_boxes = []\n\nfor index, row in train_image_df.iterrows():\n    num_boxes = get_num_boxes(row)\n    label, scores, pascal_voc_boxes = get_label(row, num_boxes)\n    all_num_boxes.append(num_boxes)\n    all_labels.append(label)\n    all_scores.append(scores)\n    all_pascal_voc_boxes.append(pascal_voc_boxes)\n    all_coco_boxes.append(get_coco_format(row))\n\ntrain_image_df['image_level'] = all_labels\ntrain_image_df['confidence_scores'] = all_scores\ntrain_image_df['pascal_voc_boxes'] = all_pascal_voc_boxes\ntrain_image_df['coco_boxes'] = all_coco_boxes\ntrain_image_df['num_boxes'] = all_num_boxes","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:43.191609Z","iopub.execute_input":"2021-09-22T12:05:43.192502Z","iopub.status.idle":"2021-09-22T12:05:44.604202Z","shell.execute_reply.started":"2021-09-22T12:05:43.192421Z","shell.execute_reply":"2021-09-22T12:05:44.603413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:44.605435Z","iopub.execute_input":"2021-09-22T12:05:44.606000Z","iopub.status.idle":"2021-09-22T12:05:44.632742Z","shell.execute_reply.started":"2021-09-22T12:05:44.605955Z","shell.execute_reply":"2021-09-22T12:05:44.631729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_df = train_image_df.drop(columns=['label'])","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:44.634213Z","iopub.execute_input":"2021-09-22T12:05:44.634828Z","iopub.status.idle":"2021-09-22T12:05:44.651240Z","shell.execute_reply.started":"2021-09-22T12:05:44.634784Z","shell.execute_reply":"2021-09-22T12:05:44.650378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"add path of dicom file to df","metadata":{}},{"cell_type":"code","source":"def get_img_id(path):\n    return path.split('/')[-1].split('.')[0] # extract img_id from path\n\ndef get_imgs_paths(root_dir):\n    paths = {}\n    for root, d_names, f_names in os.walk(root_dir):\n        for f in f_names:\n            img_id = get_img_id(os.path.join(root, f))\n            paths[img_id] = os.path.join(root, f)\n    return paths","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:44.652786Z","iopub.execute_input":"2021-09-22T12:05:44.653450Z","iopub.status.idle":"2021-09-22T12:05:44.661672Z","shell.execute_reply.started":"2021-09-22T12:05:44.653364Z","shell.execute_reply":"2021-09-22T12:05:44.660750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"paths = get_imgs_paths(data_path)\ntrain_image_df['dicom_path'] = np.nan\nfor img_id, path in paths.items():\n    train_image_df.loc[train_image_df['img_id'] == img_id, 'dicom_path'] = path","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:05:44.662887Z","iopub.execute_input":"2021-09-22T12:05:44.663207Z","iopub.status.idle":"2021-09-22T12:06:31.500978Z","shell.execute_reply.started":"2021-09-22T12:05:44.663145Z","shell.execute_reply":"2021-09-22T12:06:31.499757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:31.502688Z","iopub.execute_input":"2021-09-22T12:06:31.503101Z","iopub.status.idle":"2021-09-22T12:06:31.530845Z","shell.execute_reply.started":"2021-09-22T12:06:31.503054Z","shell.execute_reply":"2021-09-22T12:06:31.529566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-two-three\"></a>\n#### **Merge study and image levels**","metadata":{}},{"cell_type":"code","source":"# merge image and study data\ntrain_df = train_image_df.merge(train_study_df, left_on=\"StudyInstanceUID\", right_on=\"study_id\")","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:31.532448Z","iopub.execute_input":"2021-09-22T12:06:31.532903Z","iopub.status.idle":"2021-09-22T12:06:31.564691Z","shell.execute_reply.started":"2021-09-22T12:06:31.532847Z","shell.execute_reply":"2021-09-22T12:06:31.563609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:31.569804Z","iopub.execute_input":"2021-09-22T12:06:31.570149Z","iopub.status.idle":"2021-09-22T12:06:31.602321Z","shell.execute_reply.started":"2021-09-22T12:06:31.570115Z","shell.execute_reply":"2021-09-22T12:06:31.601135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(columns='study_id')\ntrain_df.head()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-09-22T12:06:31.605551Z","iopub.execute_input":"2021-09-22T12:06:31.605987Z","iopub.status.idle":"2021-09-22T12:06:31.643829Z","shell.execute_reply.started":"2021-09-22T12:06:31.605951Z","shell.execute_reply":"2021-09-22T12:06:31.642810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three\"></a>\n## **Data Analysis**","metadata":{}},{"cell_type":"code","source":"# helper function to plot frequencies\ndef plot_frequency(ax, counts_dict, title, xlabel, ylabel, xgap=0, ygap=50):\n    ax.bar(list(counts_dict.keys()), list(counts_dict.values()))\n    for i, value in enumerate(counts_dict.values()):\n        ax.text(i+xgap, value+ygap, str(value), color='#267DBE', fontweight='bold')\n\n    ax.grid(axis='y', alpha=0.75)\n    ax.set_xlabel(xlabel)\n    ax.set_ylabel(ylabel)\n    ax.set_title(title)\n    ax.set_xticks(list(counts_dict.keys()))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:31.645377Z","iopub.execute_input":"2021-09-22T12:06:31.645707Z","iopub.status.idle":"2021-09-22T12:06:31.653905Z","shell.execute_reply.started":"2021-09-22T12:06:31.645675Z","shell.execute_reply":"2021-09-22T12:06:31.652702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three-one\"></a>\n#### **Null values**","metadata":{}},{"cell_type":"code","source":"train_df.isnull().sum()","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-09-22T12:06:31.655353Z","iopub.execute_input":"2021-09-22T12:06:31.655655Z","iopub.status.idle":"2021-09-22T12:06:31.678185Z","shell.execute_reply.started":"2021-09-22T12:06:31.655626Z","shell.execute_reply":"2021-09-22T12:06:31.676945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three-two\"></a>\n#### **Duplicate values**","metadata":{}},{"cell_type":"code","source":"print(\"StudyInstanceUID is unique? {}\".format(train_df['StudyInstanceUID'].is_unique))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:31.679709Z","iopub.execute_input":"2021-09-22T12:06:31.680064Z","iopub.status.idle":"2021-09-22T12:06:31.694896Z","shell.execute_reply.started":"2021-09-22T12:06:31.680031Z","shell.execute_reply":"2021-09-22T12:06:31.693511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three-three\"></a>\n#### **Number of images per study**","metadata":{}},{"cell_type":"code","source":"study_images_count = {}\n\nfor num_images in range(1, max(train_df.StudyInstanceUID.value_counts().values)+1):\n    study_images_count[num_images] = np.count_nonzero(train_df.StudyInstanceUID.value_counts().values == num_images)\n    \ntitle = \"Number images per study\"\nxlabel = \"num images\"\nylabel = \"num studies\"\nfig, ax = plt.subplots(figsize=(8,6))\nplot_frequency(ax, study_images_count, title, xlabel, ylabel, xgap=0.8)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-09-22T12:06:31.696540Z","iopub.execute_input":"2021-09-22T12:06:31.697255Z","iopub.status.idle":"2021-09-22T12:06:31.998645Z","shell.execute_reply.started":"2021-09-22T12:06:31.697207Z","shell.execute_reply":"2021-09-22T12:06:31.997708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three-four\"></a>\n#### **Number of bboxes per image**","metadata":{}},{"cell_type":"code","source":"boxes_count = {}\nfor num_boxes in range(max(train_image_df['num_boxes'].unique())+1):\n    if num_boxes not in train_image_df['num_boxes'].unique():\n        boxes_count[num_boxes] = 0\n        continue\n    boxes_count[num_boxes] = len(train_image_df[train_image_df['num_boxes'] == num_boxes])\n\ntitle = \"Number of boxes per image\"\nxlabel = \"num boxes\"\nylabel = \"num images\"\nfig, ax = plt.subplots(figsize=(8,6))\nplot_frequency(ax, boxes_count, title, xlabel, ylabel, xgap=-0.2)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:31.999940Z","iopub.execute_input":"2021-09-22T12:06:32.000397Z","iopub.status.idle":"2021-09-22T12:06:32.231123Z","shell.execute_reply.started":"2021-09-22T12:06:32.000353Z","shell.execute_reply":"2021-09-22T12:06:32.230388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see the data contains 2040 null boxes, wich should mean image label is none and consider more than 3 bboxes as outliers","metadata":{}},{"cell_type":"code","source":"# sanity check\nnull_boxes = train_df[train_df['boxes'].isna()]\nnone_labels = train_df[train_df['image_level'] == NONE]\nprint(\"There are {} null boxes and {} none labels\".format(len(null_boxes), len(none_labels)))\nprint(\"Are all null boxes with none labels? {}\".format(len(np.intersect1d(train_image_df.loc[train_image_df['boxes'].isna(), 'img_id'].values, train_image_df.loc[train_image_df['image_level'] == NONE, 'img_id'].values)) == len(null_boxes.values)))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:32.232704Z","iopub.execute_input":"2021-09-22T12:06:32.233417Z","iopub.status.idle":"2021-09-22T12:06:32.264787Z","shell.execute_reply.started":"2021-09-22T12:06:32.233370Z","shell.execute_reply":"2021-09-22T12:06:32.263456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three-five\"></a>\n#### **Study-level class frequency**","metadata":{}},{"cell_type":"code","source":"# check data sparsity for study\nnum_negatives = len(train_df[train_df[NEGATIVE] == 1])\nnum_typicals = len(train_df[train_df[TYPICAL] == 1])\nnum_indeterminates = len(train_df[train_df[INDERTEMINATE] == 1])\nnum_atypicals = len(train_df[train_df[ATYPICAL] == 1])\n\nstudy_labels_count = {NEGATIVE:num_negatives, TYPICAL: num_typicals, \n                      INDERTEMINATE: num_indeterminates, ATYPICAL: num_atypicals}\ntitle = \"Number of studies per label\"\nxlabel = \"study-level labels\"\nylabel = \"number of images\"\nfig, ax = plt.subplots(figsize=(8,6))\n\nplot_frequency(ax, study_labels_count, title, xlabel, ylabel, xgap=-0.1)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:32.266288Z","iopub.execute_input":"2021-09-22T12:06:32.266712Z","iopub.status.idle":"2021-09-22T12:06:32.444067Z","shell.execute_reply.started":"2021-09-22T12:06:32.266668Z","shell.execute_reply":"2021-09-22T12:06:32.443090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three-six\"></a>\n#### **Image-level class frequency**","metadata":{}},{"cell_type":"code","source":"# check data sparsity \nnum_none = len(train_image_df[train_df['image_level'] == NONE])\nnum_opacity = len(train_image_df[train_df['image_level'] == OPACITY])\n\nstudy_labels_count = {NONE:num_none, OPACITY: num_opacity}\n\ntitle = \"Number of images per label\"\nxlabel = \"num samples\"\nylabel = \"image-level labels\"\nfig, ax = plt.subplots(figsize=(6,5))\nplot_frequency(ax, study_labels_count, title, xlabel, ylabel, xgap=-0.1)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-09-22T12:06:32.445262Z","iopub.execute_input":"2021-09-22T12:06:32.445713Z","iopub.status.idle":"2021-09-22T12:06:32.583592Z","shell.execute_reply.started":"2021-09-22T12:06:32.445669Z","shell.execute_reply":"2021-09-22T12:06:32.582181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three-seven\"></a>\n#### **Class distribution with no bbox**","metadata":{}},{"cell_type":"code","source":"labels_null_boxes_count = {}\n\nfor label in study_level_labels:\n    labels_null_boxes_count[label] = len(train_df[((train_df['boxes'].isna()) & (train_df[label]==1))])\n    \ntitle = \"Number of images with null boxes per study-level labels\"\nxlabel = \"study-level labels\"\nylabel = \"num images with null boxes\"\nfig, ax = plt.subplots(figsize=(6,6))\nplot_frequency(ax, labels_null_boxes_count, title, xlabel, ylabel, xgap=-0.1, ygap=20)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:32.584993Z","iopub.execute_input":"2021-09-22T12:06:32.585317Z","iopub.status.idle":"2021-09-22T12:06:32.778762Z","shell.execute_reply.started":"2021-09-22T12:06:32.585285Z","shell.execute_reply":"2021-09-22T12:06:32.777710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three-eight\"></a>\n#### **Number of boxes for the different classes**","metadata":{}},{"cell_type":"code","source":"negatives_boxes_count = {}\ntypicals_boxes_count = {}\natypicals_boxes_count = {}\ninderteminates_boxes_count = {}\n\nfor label in study_level_labels:\n    for num_boxes in range(max(train_image_df['num_boxes'].unique())+1):\n        if label == NEGATIVE:\n            negatives_boxes_count[num_boxes] = len(train_df[((train_df['num_boxes'] == num_boxes) & (train_df[NEGATIVE] == 1))])\n        if label == TYPICAL:\n            typicals_boxes_count[num_boxes] = len(train_df[((train_df['num_boxes'] == num_boxes) & (train_df[TYPICAL] == 1))])\n        if label == ATYPICAL:\n            atypicals_boxes_count[num_boxes] = len(train_df[((train_df['num_boxes'] == num_boxes) & (train_df[ATYPICAL] == 1))])\n        else:\n            inderteminates_boxes_count[num_boxes] = len(train_df[((train_df['num_boxes'] == num_boxes) & (train_df[INDERTEMINATE] == 1))])\n\n            \n            \nfig, axes = plt.subplots(nrows=2, ncols=2, figsize=(20,10))\nplot_frequency(axes[0,0], negatives_boxes_count, \"Box conuting for negatives\", \"num boxes\", \"num samples\", -0.1, 20)\nplot_frequency(axes[0,1], typicals_boxes_count, \"Box conuting for typicals\", \"num boxes\", \"num samples\", -0.2, 20)\nplot_frequency(axes[1,0], atypicals_boxes_count, \"Box conuting for atypicals\", \"num boxes\", \"num samples\", -0.1, 5)\nplot_frequency(axes[1,1], inderteminates_boxes_count, \"Box conuting for inderteminates\", \"num boxes\", \"num samples\", -0.1, 10)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:32.780441Z","iopub.execute_input":"2021-09-22T12:06:32.780811Z","iopub.status.idle":"2021-09-22T12:06:33.767763Z","shell.execute_reply.started":"2021-09-22T12:06:32.780776Z","shell.execute_reply":"2021-09-22T12:06:33.766765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-three-nine\"></a>\n#### **Relation between box size and number of boxes per image**","metadata":{}},{"cell_type":"code","source":"num_boxes = np.arange(1, max(train_image_df['num_boxes'].unique())+1)\nboxes_size = {}\n\nfor key in num_boxes:\n    boxes_size[key] = []\n\nfor index, row in train_image_df.iterrows():\n    num_boxes = row['num_boxes']\n    if num_boxes != 0:\n        boxes = row['coco_boxes']\n        for box in boxes:\n            x,y,w,h = box\n            size = w*h\n            boxes_size[num_boxes].append(size)\n\nfig, ax = plt.subplots()\nfor num_boxes, boxes_size in boxes_size.items():\n    ax.scatter([num_boxes]*len(boxes_size), boxes_size, label=num_boxes)\n\nax.ticklabel_format(style='plain', useOffset=False)\nplt.xlabel('num of boxes')\nplt.ylabel('sizes')    \nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:33.769181Z","iopub.execute_input":"2021-09-22T12:06:33.769524Z","iopub.status.idle":"2021-09-22T12:06:35.134017Z","shell.execute_reply.started":"2021-09-22T12:06:33.769481Z","shell.execute_reply":"2021-09-22T12:06:35.132954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We understand that for num_boxes > 4 the boxes size are very small and those are outliers","metadata":{}},{"cell_type":"code","source":"train_df = train_df[train_df['num_boxes'] < 4]","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:35.135465Z","iopub.execute_input":"2021-09-22T12:06:35.135774Z","iopub.status.idle":"2021-09-22T12:06:35.143011Z","shell.execute_reply.started":"2021-09-22T12:06:35.135739Z","shell.execute_reply":"2021-09-22T12:06:35.141908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-four\"></a>\n## **DICOM files**","metadata":{}},{"cell_type":"markdown","source":"<a id=\"section-four-one\"></a>\n#### **View dicom files metadata**","metadata":{}},{"cell_type":"code","source":"pydicom.dcmread(train_df.loc[0, 'dicom_path'])","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:35.144586Z","iopub.execute_input":"2021-09-22T12:06:35.145018Z","iopub.status.idle":"2021-09-22T12:06:36.096710Z","shell.execute_reply.started":"2021-09-22T12:06:35.144971Z","shell.execute_reply":"2021-09-22T12:06:36.095663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-four-two\"></a>\n#### **Add metadata and image shape to train df**","metadata":{}},{"cell_type":"code","source":"def get_img(path):\n        data_file = pydicom.dcmread(path)\n        img = apply_voi_lut(data_file.pixel_array, data_file)\n        #img = data_file.pixel_array.astype(float)\n\n        if data_file.PhotometricInterpretation == \"MONOCHROME1\":\n            img = np.amax(img) - img\n\n        # Rescaling grey scale between 0-255 and convert to uint\n        img = img - np.min(img)\n        img = img / np.max(img)\n        img = (img * 255).astype(np.uint8)\n\n        return img\n\ndef get_img_id(path):\n        return path.split('/')[-1].split('.')[0] # extract img_id from path","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:36.098084Z","iopub.execute_input":"2021-09-22T12:06:36.098425Z","iopub.status.idle":"2021-09-22T12:06:36.105749Z","shell.execute_reply.started":"2021-09-22T12:06:36.098393Z","shell.execute_reply":"2021-09-22T12:06:36.104615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_observation_data(path):\n    image_data = pydicom.read_file(path)\n    img_id = get_img_id(path)\n    \n    # Dictionary to store the information from the image\n    observation_data = {\n        \"img_id\": img_id,\n        \"Rows\" : image_data.get(\"Rows\"),\n        \"Columns\" : image_data.get(\"Columns\"),\n        \"SOPClassUID\" : image_data.get(\"SOPClassUID\"),\n        \"SOPInstanceUID\" : image_data.get(\"SOPInstanceUID\"),\n        \"PatientID\" : image_data.get(\"PatientID\"),\n        \"PatientName\" : image_data.get(\"PatientName\"),\n        \"PatientSex\" : image_data.get(\"PatientSex\"),\n        \"PhotometricInterpretation\" : image_data.get(\"PhotometricInterpretation\"),\n        \"StudyInstanceUID\" : image_data.get(\"StudyInstanceUID\"),\n        \"SamplesPerPixel\" : image_data.get(\"SamplesPerPixel\"),\n        \"BitsAllocated\" : image_data.get(\"BitsAllocated\"),\n        \"BitsStored\" : image_data.get(\"BitsStored\"),\n        \"HighBit\" : image_data.get(\"HighBit\"),\n        \"PixelRepresentation\" : image_data.get(\"PixelRepresentation\"),\n    }\n\n    # String columns\n    str_columns = [\"ImageType\", \"Modality\", \"PatientSex\", \"BodyPartExamined\"]\n    for i in str_columns:\n        observation_data[i] = str(image_data.get(i)) if i in image_data else None\n        \n    return observation_data","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:36.107746Z","iopub.execute_input":"2021-09-22T12:06:36.108231Z","iopub.status.idle":"2021-09-22T12:06:36.120334Z","shell.execute_reply.started":"2021-09-22T12:06:36.108183Z","shell.execute_reply":"2021-09-22T12:06:36.119329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata = {}\nshapes = []\n\nfor index, row in train_df.iterrows():\n    metadata[index] = get_observation_data(row['dicom_path'])\n    img = get_img(row['dicom_path'])\n    shapes.append(img.shape)\n\ntrain_df['image_shape'] = shapes","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:06:36.122115Z","iopub.execute_input":"2021-09-22T12:06:36.122607Z","iopub.status.idle":"2021-09-22T12:40:40.229070Z","shell.execute_reply.started":"2021-09-22T12:06:36.122572Z","shell.execute_reply":"2021-09-22T12:40:40.219174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df = pd.DataFrame(metadata)\n# swap the columns with indexes\nmetadata_df = metadata_df.transpose()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:40:40.247727Z","iopub.execute_input":"2021-09-22T12:40:40.248266Z","iopub.status.idle":"2021-09-22T12:40:40.961044Z","shell.execute_reply.started":"2021-09-22T12:40:40.248197Z","shell.execute_reply":"2021-09-22T12:40:40.960137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:40:40.962624Z","iopub.execute_input":"2021-09-22T12:40:40.963215Z","iopub.status.idle":"2021-09-22T12:40:41.012619Z","shell.execute_reply.started":"2021-09-22T12:40:40.963149Z","shell.execute_reply":"2021-09-22T12:40:41.011580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# verify read meta data of all images\ntrain_ids = np.array(train_df['img_id'].values) \nmetadata_ids = np.array(metadata_df['img_id'].values)\nlen(np.setdiff1d(train_ids,metadata_ids))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:40:41.014136Z","iopub.execute_input":"2021-09-22T12:40:41.014777Z","iopub.status.idle":"2021-09-22T12:40:43.176437Z","shell.execute_reply.started":"2021-09-22T12:40:41.014729Z","shell.execute_reply":"2021-09-22T12:40:43.175416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df.to_csv(os.path.join(output_path, 'images_metadata.csv'), index=False)\n#metadata_df = pd.read_csv(metadata_path)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:40:43.177896Z","iopub.execute_input":"2021-09-22T12:40:43.178371Z","iopub.status.idle":"2021-09-22T12:40:43.316543Z","shell.execute_reply.started":"2021-09-22T12:40:43.178317Z","shell.execute_reply":"2021-09-22T12:40:43.315536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# merge metadata and train df\ntrain_df = pd.merge(train_df, metadata_df, how='inner', on=['img_id'])\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:40:43.318074Z","iopub.execute_input":"2021-09-22T12:40:43.318419Z","iopub.status.idle":"2021-09-22T12:40:43.418492Z","shell.execute_reply.started":"2021-09-22T12:40:43.318386Z","shell.execute_reply":"2021-09-22T12:40:43.417482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(columns=['StudyInstanceUID_y']).rename(columns = {'StudyInstanceUID_x': 'StudyInstanceUID'})\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:40:43.419830Z","iopub.execute_input":"2021-09-22T12:40:43.420171Z","iopub.status.idle":"2021-09-22T12:40:43.479232Z","shell.execute_reply.started":"2021-09-22T12:40:43.420128Z","shell.execute_reply":"2021-09-22T12:40:43.478188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-five\"></a>\n## **Explore images**","metadata":{}},{"cell_type":"code","source":"def get_study_label(sample):\n    if sample[NEGATIVE].values[0] == 1:\n        return NEGATIVE\n    if sample[TYPICAL].values[0] == 1:\n        return TYPICAL\n    if sample[INDERTEMINATE].values[0] == 1:\n        return INDERTEMINATE\n    return ATYPICAL","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:40:43.480709Z","iopub.execute_input":"2021-09-22T12:40:43.481032Z","iopub.status.idle":"2021-09-22T12:40:43.488460Z","shell.execute_reply.started":"2021-09-22T12:40:43.481003Z","shell.execute_reply":"2021-09-22T12:40:43.487182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"studies_per_imgs_count = {}\nimgs_paths = {}\nimgs_count = train_df.StudyInstanceUID.value_counts()\n\nfor num_images in range(2,10):\n    studies_per_imgs_count[num_images] = imgs_count.where(imgs_count == num_images).dropna().keys()\n\nfor num_images, studies in studies_per_imgs_count.items():\n    study_paths = {}\n    for study in studies:\n        paths = []\n        for root, d_names, f_names in os.walk(os.path.join(train_path,study)):\n            for f in f_names:\n                studies_per_imgs_count[num_images]\n                paths.append(os.path.join(root, f))\n        study_paths[study] = paths\n        \n    imgs_paths[num_images] = study_paths","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:40:43.489913Z","iopub.execute_input":"2021-09-22T12:40:43.490223Z","iopub.status.idle":"2021-09-22T12:40:45.050122Z","shell.execute_reply.started":"2021-09-22T12:40:43.490187Z","shell.execute_reply":"2021-09-22T12:40:45.049138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_studies_imgs_by_num_img(df, studies, num_images, figsize):\n    fig, axes = plt.subplots(nrows=len(studies), ncols=num_images, figsize=figsize) \n    colors = {TYPICAL: (0,0,255), INDERTEMINATE: (0,255,0), ATYPICAL: (255,0,0)} # negatives have no boxes\n    print(\"Typical: \"+colored(\"Blue\",\"blue\")+\"\\nInderteminate: \"+colored(\"Green\",\"green\")+\"\\nAtypical: \"+colored(\"Red\", \"red\"))\n            \n    for row, (study, paths) in enumerate(studies.items()):\n        for col, path in enumerate(paths):\n            img_id = get_img_id(path)\n            img = get_img(path)\n            # create new RGB image from original\n            new_img = np.zeros((img.shape[0], img.shape[1], 3), dtype=img.dtype)\n            new_img[:,:,:] = img[:,:,np.newaxis]\n            row_df = df[df['img_id'] == img_id]\n            study_label = get_study_label(row_df)\n            \n            if row_df['image_level'].values[0] != NONE:\n                for box in df['pascal_voc_boxes'].values[0]:\n                    xmin, ymin, xmax, ymax = int(box[0]), int(box[1]), int(box[2]), int(box[3])\n                    new_img = cv2.rectangle(new_img,(xmin,ymin),(xmax,ymax),colors[study_label],20)\n\n            if len(studies) > 1:\n                axes[row,col].set_title(\"Study: {}\\nImage ID: {}\\nLabel: {}\".format(study, img_id, study_label))\n                axes[row,col].imshow(new_img)\n            else:\n                axes[col].set_title(\"Study: {}\\nImage ID: {}\\nLabel: {}\".format(study, img_id, study_label))\n                axes[col].imshow(new_img)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:40:45.051786Z","iopub.execute_input":"2021-09-22T12:40:45.052110Z","iopub.status.idle":"2021-09-22T12:40:45.068415Z","shell.execute_reply.started":"2021-09-22T12:40:45.052079Z","shell.execute_reply":"2021-09-22T12:40:45.067338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-five-one\"></a>\n#### **Number of images per study**","metadata":{}},{"cell_type":"code","source":"study_count = {}\ntotal_with_more_than_one = 0\n\nfor num_imgs,studies in imgs_paths.items():\n    study_count[num_imgs] = len(studies)\n    total_with_more_than_one += len(studies)\n\nstudy_count[1] = len(train_df) - total_with_more_than_one\n\n# sort dict by count\nstudy_count = dict(sorted(study_count.items()))\n\ntitle = \"Number images per study\"\nxlabel = \"num images\"\nylabel = \"num studies\"\nfig, ax = plt.subplots(figsize=(8,6))\nplot_frequency(ax, study_count, title, xlabel, ylabel, xgap=0.8)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:40:45.069938Z","iopub.execute_input":"2021-09-22T12:40:45.070699Z","iopub.status.idle":"2021-09-22T12:40:45.374099Z","shell.execute_reply.started":"2021-09-22T12:40:45.070662Z","shell.execute_reply":"2021-09-22T12:40:45.372999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's explore studies with more than 1 image","metadata":{}},{"cell_type":"markdown","source":"<a id=\"section-five-two\"></a>\n#### **Studies with 3 images**","metadata":{}},{"cell_type":"code","source":"show_studies_imgs_by_num_img(train_df, imgs_paths[3], 3, (15,100))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:40:45.375478Z","iopub.execute_input":"2021-09-22T12:40:45.375941Z","iopub.status.idle":"2021-09-22T12:41:45.890664Z","shell.execute_reply.started":"2021-09-22T12:40:45.375906Z","shell.execute_reply":"2021-09-22T12:41:45.889228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see in each study the images are the same and those wh have an image with a bounding box, just one of the images contains it, even though it's the same image. </br>\nRemove images without bbox","metadata":{}},{"cell_type":"code","source":"imgs_paths[3].keys()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:41:45.892397Z","iopub.execute_input":"2021-09-22T12:41:45.892794Z","iopub.status.idle":"2021-09-22T12:41:45.899120Z","shell.execute_reply.started":"2021-09-22T12:41:45.892757Z","shell.execute_reply":"2021-09-22T12:41:45.898221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"studies_without_bbox = ['e764f1cb364c', '0d9709b3af74', '7416b5cbc531']\nstudies_with_bbox = [study for study in list(imgs_paths[3].keys()) if study not in studies_without_bbox]\n\nprint(len(list(imgs_paths[3].keys())) == (len(studies_with_bbox)+len(studies_without_bbox)))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:41:45.901114Z","iopub.execute_input":"2021-09-22T12:41:45.901576Z","iopub.status.idle":"2021-09-22T12:41:45.910319Z","shell.execute_reply.started":"2021-09-22T12:41:45.901536Z","shell.execute_reply":"2021-09-22T12:41:45.909141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def compare_columns(studies, num_imgs):\n    study_imgs = {}\n\n    for study in studies:\n        rows = train_df[train_df['StudyInstanceUID']==study]\n        study_imgs[study] = rows.to_dict(orient='records')\n\n    for study, samples in study_imgs.items():\n        print(\"\\033[1mStudy: {}\\n\\033[0m\".format(study))\n        for key in list(samples[0].keys()):\n            values = [samples[i][key] for i in range(num_imgs)]\n            print(\"\\033[1m{}:\\033[0m\".format(key))\n            for value in values:\n                print(value)\n            print()\n        print(\"------------------------------------------------------------------------------------------------------------------------------------------\")\n    \ncompare_columns(studies_with_bbox, 3)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:41:45.911891Z","iopub.execute_input":"2021-09-22T12:41:45.912581Z","iopub.status.idle":"2021-09-22T12:41:46.164827Z","shell.execute_reply.started":"2021-09-22T12:41:45.912539Z","shell.execute_reply":"2021-09-22T12:41:46.163940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see all images of the same study has the same PatientID and PatientName and is the same scan, so we choose to keep only the one image with the bbox.","metadata":{}},{"cell_type":"code","source":"def drop_imgs_without_bbox(studies, df):\n    rows_to_drop = []\n    for study in studies:\n        rows = df[df['StudyInstanceUID']==study]\n        for row in rows.loc[rows['num_boxes']==0].index:\n            rows_to_drop.append(row)\n    \n    return df.drop(labels=rows_to_drop, axis=0)\n\ntrain_df = drop_imgs_without_bbox(studies_with_bbox, train_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:41:46.166399Z","iopub.execute_input":"2021-09-22T12:41:46.166831Z","iopub.status.idle":"2021-09-22T12:41:46.205275Z","shell.execute_reply.started":"2021-09-22T12:41:46.166789Z","shell.execute_reply":"2021-09-22T12:41:46.204227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check the differences between images of same study without bounding box","metadata":{}},{"cell_type":"code","source":"compare_columns(studies_without_bbox, 3)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:41:46.206434Z","iopub.execute_input":"2021-09-22T12:41:46.206703Z","iopub.status.idle":"2021-09-22T12:41:46.271470Z","shell.execute_reply.started":"2021-09-22T12:41:46.206676Z","shell.execute_reply":"2021-09-22T12:41:46.267245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"they are all duplicates, we choose to keep only one of them","metadata":{}},{"cell_type":"code","source":"def drop_duplicate_imgs(studies, df, num_imgs):\n    rows_to_drop = []\n    for study in studies:\n        rows = df[df['StudyInstanceUID']==study]\n        for i, row in enumerate(rows.index):\n            if i%num_imgs!=0:\n                rows_to_drop.append(row)\n                \n    return df.drop(labels=rows_to_drop, axis=0)\n\ntrain_df = drop_duplicate_imgs(studies_without_bbox, train_df, 3)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:41:46.272713Z","iopub.execute_input":"2021-09-22T12:41:46.272989Z","iopub.status.idle":"2021-09-22T12:41:46.290903Z","shell.execute_reply.started":"2021-09-22T12:41:46.272962Z","shell.execute_reply":"2021-09-22T12:41:46.289641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see all columns have same values, except num_boxes","metadata":{}},{"cell_type":"markdown","source":"<a id=\"section-five-three\"></a>\n#### **Studies with 4 images**","metadata":{}},{"cell_type":"code","source":"show_studies_imgs_by_num_img(train_df, imgs_paths[4], 4, (20,20))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:41:46.292626Z","iopub.execute_input":"2021-09-22T12:41:46.292944Z","iopub.status.idle":"2021-09-22T12:42:08.989252Z","shell.execute_reply.started":"2021-09-22T12:41:46.292908Z","shell.execute_reply":"2021-09-22T12:42:08.987687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs_paths[4].keys()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:08.990711Z","iopub.execute_input":"2021-09-22T12:42:08.991012Z","iopub.status.idle":"2021-09-22T12:42:08.996554Z","shell.execute_reply.started":"2021-09-22T12:42:08.990983Z","shell.execute_reply":"2021-09-22T12:42:08.995620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"studies_without_bbox = ['74ba8f2badcb']\nstudies_with_bbox = [study for study in list(imgs_paths[4].keys()) if study not in studies_without_bbox]\n\nprint(len(list(imgs_paths[4].keys())) == (len(studies_with_bbox)+len(studies_without_bbox)))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:08.997918Z","iopub.execute_input":"2021-09-22T12:42:08.998305Z","iopub.status.idle":"2021-09-22T12:42:09.008456Z","shell.execute_reply.started":"2021-09-22T12:42:08.998271Z","shell.execute_reply":"2021-09-22T12:42:09.007264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"compare_columns(studies_with_bbox, 4)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:09.009940Z","iopub.execute_input":"2021-09-22T12:42:09.010246Z","iopub.status.idle":"2021-09-22T12:42:09.115280Z","shell.execute_reply.started":"2021-09-22T12:42:09.010216Z","shell.execute_reply":"2021-09-22T12:42:09.114128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"like before, for each study we keep only one image- the one with the bbox","metadata":{}},{"cell_type":"code","source":"train_df = drop_imgs_without_bbox(studies_with_bbox, train_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:09.116762Z","iopub.execute_input":"2021-09-22T12:42:09.117115Z","iopub.status.idle":"2021-09-22T12:42:09.136647Z","shell.execute_reply.started":"2021-09-22T12:42:09.117073Z","shell.execute_reply":"2021-09-22T12:42:09.135665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check the differences between images of same study without bounding box","metadata":{}},{"cell_type":"code","source":"compare_columns(studies_without_bbox, 4)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:09.137998Z","iopub.execute_input":"2021-09-22T12:42:09.138333Z","iopub.status.idle":"2021-09-22T12:42:09.168530Z","shell.execute_reply.started":"2021-09-22T12:42:09.138287Z","shell.execute_reply":"2021-09-22T12:42:09.166964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Again, the images are duplicate- keep only one image per study","metadata":{}},{"cell_type":"code","source":"train_df = drop_duplicate_imgs(studies_without_bbox, train_df, 4)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:09.170679Z","iopub.execute_input":"2021-09-22T12:42:09.171124Z","iopub.status.idle":"2021-09-22T12:42:09.190427Z","shell.execute_reply.started":"2021-09-22T12:42:09.171080Z","shell.execute_reply":"2021-09-22T12:42:09.189510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-five-four\"></a>\n#### **Studies with 5 images**","metadata":{}},{"cell_type":"code","source":"show_studies_imgs_by_num_img(train_df, imgs_paths[5], 5, (20,20))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:09.191746Z","iopub.execute_input":"2021-09-22T12:42:09.192063Z","iopub.status.idle":"2021-09-22T12:42:25.190805Z","shell.execute_reply.started":"2021-09-22T12:42:09.192020Z","shell.execute_reply":"2021-09-22T12:42:25.189742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs_paths[5].keys()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:25.192345Z","iopub.execute_input":"2021-09-22T12:42:25.192748Z","iopub.status.idle":"2021-09-22T12:42:25.198851Z","shell.execute_reply.started":"2021-09-22T12:42:25.192706Z","shell.execute_reply":"2021-09-22T12:42:25.197756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"studies_without_bbox = ['a0254bf8a96e']\nstudies_with_bbox = [study for study in list(imgs_paths[5].keys()) if study not in studies_without_bbox]\n\nprint(len(list(imgs_paths[5].keys())) == (len(studies_with_bbox)+len(studies_without_bbox)))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:25.200604Z","iopub.execute_input":"2021-09-22T12:42:25.200986Z","iopub.status.idle":"2021-09-22T12:42:25.209289Z","shell.execute_reply.started":"2021-09-22T12:42:25.200947Z","shell.execute_reply":"2021-09-22T12:42:25.208309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"compare_columns(studies_with_bbox, 5)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:25.217664Z","iopub.execute_input":"2021-09-22T12:42:25.217941Z","iopub.status.idle":"2021-09-22T12:42:25.296854Z","shell.execute_reply.started":"2021-09-22T12:42:25.217917Z","shell.execute_reply":"2021-09-22T12:42:25.272249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = drop_imgs_without_bbox(studies_with_bbox, train_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:25.299749Z","iopub.execute_input":"2021-09-22T12:42:25.300141Z","iopub.status.idle":"2021-09-22T12:42:25.314902Z","shell.execute_reply.started":"2021-09-22T12:42:25.300110Z","shell.execute_reply":"2021-09-22T12:42:25.314027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"compare_columns(studies_without_bbox, 5)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:25.316112Z","iopub.execute_input":"2021-09-22T12:42:25.316506Z","iopub.status.idle":"2021-09-22T12:42:25.348288Z","shell.execute_reply.started":"2021-09-22T12:42:25.316387Z","shell.execute_reply":"2021-09-22T12:42:25.341688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = drop_duplicate_imgs(studies_without_bbox, train_df, 5)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:25.349772Z","iopub.execute_input":"2021-09-22T12:42:25.350229Z","iopub.status.idle":"2021-09-22T12:42:25.363257Z","shell.execute_reply.started":"2021-09-22T12:42:25.350187Z","shell.execute_reply":"2021-09-22T12:42:25.362244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-five-five\"></a>\n#### **Studies with 6 images**","metadata":{}},{"cell_type":"code","source":"show_studies_imgs_by_num_img(train_df, imgs_paths[6], 6, (20,20))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:25.364805Z","iopub.execute_input":"2021-09-22T12:42:25.365527Z","iopub.status.idle":"2021-09-22T12:42:30.208139Z","shell.execute_reply.started":"2021-09-22T12:42:25.365483Z","shell.execute_reply":"2021-09-22T12:42:30.207199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs_paths[6].keys()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:30.209403Z","iopub.execute_input":"2021-09-22T12:42:30.209692Z","iopub.status.idle":"2021-09-22T12:42:30.214594Z","shell.execute_reply.started":"2021-09-22T12:42:30.209662Z","shell.execute_reply":"2021-09-22T12:42:30.213889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"studies_with_bbox = ['8943d1d85097']\n\nprint(len(list(imgs_paths[6].keys())) == (len(studies_with_bbox)))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:30.215643Z","iopub.execute_input":"2021-09-22T12:42:30.216022Z","iopub.status.idle":"2021-09-22T12:42:30.226779Z","shell.execute_reply.started":"2021-09-22T12:42:30.215994Z","shell.execute_reply":"2021-09-22T12:42:30.226069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"compare_columns(studies_with_bbox, 6)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:30.227822Z","iopub.execute_input":"2021-09-22T12:42:30.228228Z","iopub.status.idle":"2021-09-22T12:42:30.269936Z","shell.execute_reply.started":"2021-09-22T12:42:30.228173Z","shell.execute_reply":"2021-09-22T12:42:30.267178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = drop_imgs_without_bbox(studies_with_bbox, train_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:30.271413Z","iopub.execute_input":"2021-09-22T12:42:30.271835Z","iopub.status.idle":"2021-09-22T12:42:30.285149Z","shell.execute_reply.started":"2021-09-22T12:42:30.271792Z","shell.execute_reply":"2021-09-22T12:42:30.283808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-five-six\"></a>\n#### **Studies with 7 images**","metadata":{}},{"cell_type":"code","source":"show_studies_imgs_by_num_img(train_df, imgs_paths[7], 7, (20,20))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:30.287080Z","iopub.execute_input":"2021-09-22T12:42:30.287521Z","iopub.status.idle":"2021-09-22T12:42:38.260384Z","shell.execute_reply.started":"2021-09-22T12:42:30.287475Z","shell.execute_reply":"2021-09-22T12:42:38.259132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs_paths[7].keys()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:38.262284Z","iopub.execute_input":"2021-09-22T12:42:38.262723Z","iopub.status.idle":"2021-09-22T12:42:38.270093Z","shell.execute_reply.started":"2021-09-22T12:42:38.262653Z","shell.execute_reply":"2021-09-22T12:42:38.268761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"studies_with_bbox = ['a7335b2f9815']\n\nprint(len(list(imgs_paths[7].keys())) == (len(studies_with_bbox)))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:38.271714Z","iopub.execute_input":"2021-09-22T12:42:38.272457Z","iopub.status.idle":"2021-09-22T12:42:38.279874Z","shell.execute_reply.started":"2021-09-22T12:42:38.272342Z","shell.execute_reply":"2021-09-22T12:42:38.278729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = drop_imgs_without_bbox(studies_with_bbox, train_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:38.281631Z","iopub.execute_input":"2021-09-22T12:42:38.282079Z","iopub.status.idle":"2021-09-22T12:42:38.296759Z","shell.execute_reply.started":"2021-09-22T12:42:38.282038Z","shell.execute_reply":"2021-09-22T12:42:38.295743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-five-seven\"></a>\n#### **Studies with 9 images**","metadata":{}},{"cell_type":"code","source":"show_studies_imgs_by_num_img(train_df, imgs_paths[9], 9, (40,20))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:38.298400Z","iopub.execute_input":"2021-09-22T12:42:38.298962Z","iopub.status.idle":"2021-09-22T12:42:50.155410Z","shell.execute_reply.started":"2021-09-22T12:42:38.298921Z","shell.execute_reply":"2021-09-22T12:42:50.154391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs_paths[9].keys()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:50.156786Z","iopub.execute_input":"2021-09-22T12:42:50.157071Z","iopub.status.idle":"2021-09-22T12:42:50.163418Z","shell.execute_reply.started":"2021-09-22T12:42:50.157041Z","shell.execute_reply":"2021-09-22T12:42:50.162324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"studies_with_bbox = ['0fd2db233deb']\n\nprint(len(list(imgs_paths[9].keys())) == (len(studies_with_bbox)))","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:50.164696Z","iopub.execute_input":"2021-09-22T12:42:50.164987Z","iopub.status.idle":"2021-09-22T12:42:50.173297Z","shell.execute_reply.started":"2021-09-22T12:42:50.164957Z","shell.execute_reply":"2021-09-22T12:42:50.172263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = drop_imgs_without_bbox(studies_with_bbox, train_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:50.174650Z","iopub.execute_input":"2021-09-22T12:42:50.175059Z","iopub.status.idle":"2021-09-22T12:42:50.191574Z","shell.execute_reply.started":"2021-09-22T12:42:50.175028Z","shell.execute_reply":"2021-09-22T12:42:50.190303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-five-eight\"></a>\n#### **Studies with 2 images**","metadata":{}},{"cell_type":"markdown","source":"there are 206 studies with 2 images, instead of checking each study we assume the same pattern for them- duplicates, when both don't have any bbox or one of them have","metadata":{}},{"cell_type":"code","source":"rows_to_drop = []\n\nfor study, samples in imgs_paths[2].items():\n    rows = train_df[train_df['StudyInstanceUID']==study]\n    sample1= rows.iloc[0]\n    sample2 = rows.iloc[1]\n    if sample1['PatientID'] == sample2['PatientID']:  # if same patient id (duplicate)\n        if sample1['num_boxes'] != sample2['num_boxes']: # if not same number of bounding boxes\n            rows_to_drop.append(rows.loc[rows['num_boxes']==0].index.values[0]) # keep image with bounding box\n        if sample1['num_boxes'] == sample2['num_boxes']: # if same number of boxes (probably no boxes)\n            rows_to_drop.append(rows.loc[rows['img_id']==sample1['img_id']].index.values[0]) # keep only one of them (sample2)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:50.193117Z","iopub.execute_input":"2021-09-22T12:42:50.193454Z","iopub.status.idle":"2021-09-22T12:42:50.698477Z","shell.execute_reply.started":"2021-09-22T12:42:50.193424Z","shell.execute_reply":"2021-09-22T12:42:50.697596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(labels=rows_to_drop, axis=0)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:50.699701Z","iopub.execute_input":"2021-09-22T12:42:50.699973Z","iopub.status.idle":"2021-09-22T12:42:50.709736Z","shell.execute_reply.started":"2021-09-22T12:42:50.699927Z","shell.execute_reply":"2021-09-22T12:42:50.708735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check null boxes after \"cleaning\" the data","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:50.711403Z","iopub.execute_input":"2021-09-22T12:42:50.711845Z","iopub.status.idle":"2021-09-22T12:42:50.752233Z","shell.execute_reply.started":"2021-09-22T12:42:50.711810Z","shell.execute_reply":"2021-09-22T12:42:50.751332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-five-nine\"></a>\n#### **Box count per class**","metadata":{}},{"cell_type":"code","source":"negatives_boxes_count = {}\ntypicals_boxes_count = {}\natypicals_boxes_count = {}\ninderteminates_boxes_count = {}\n\nfor label in study_level_labels:\n    for num_boxes in range(max(train_image_df['num_boxes'].unique())+1):\n        if label == NEGATIVE:\n            negatives_boxes_count[num_boxes] = len(train_df[((train_df['num_boxes'] == num_boxes) & (train_df[NEGATIVE] == 1))])\n        if label == TYPICAL:\n            typicals_boxes_count[num_boxes] = len(train_df[((train_df['num_boxes'] == num_boxes) & (train_df[TYPICAL] == 1))])\n        if label == ATYPICAL:\n            atypicals_boxes_count[num_boxes] = len(train_df[((train_df['num_boxes'] == num_boxes) & (train_df[ATYPICAL] == 1))])\n        else:\n            inderteminates_boxes_count[num_boxes] = len(train_df[((train_df['num_boxes'] == num_boxes) & (train_df[INDERTEMINATE] == 1))])\n\n            \n            \nfig, axes = plt.subplots(nrows=2, ncols=2, figsize=(20,10))\nplot_frequency(axes[0,0], negatives_boxes_count, \"Box conuting for negatives\", \"num boxes\", \"num samples\", -0.1, 20)\nplot_frequency(axes[0,1], typicals_boxes_count, \"Box conuting for typicals\", \"num boxes\", \"num samples\", -0.2, 20)\nplot_frequency(axes[1,0], atypicals_boxes_count, \"Box conuting for atypicals\", \"num boxes\", \"num samples\", -0.1, 5)\nplot_frequency(axes[1,1], inderteminates_boxes_count, \"Box conuting for inderteminates\", \"num boxes\", \"num samples\", -0.1, 10)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:50.753589Z","iopub.execute_input":"2021-09-22T12:42:50.753905Z","iopub.status.idle":"2021-09-22T12:42:51.638542Z","shell.execute_reply.started":"2021-09-22T12:42:50.753878Z","shell.execute_reply":"2021-09-22T12:42:51.637607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# remove outliers for typicals\ntrain_df = train_df.drop(index=train_df[(train_df['int_label']==1)&(train_df['num_boxes']<2)].index)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:51.639835Z","iopub.execute_input":"2021-09-22T12:42:51.640124Z","iopub.status.idle":"2021-09-22T12:42:51.650713Z","shell.execute_reply.started":"2021-09-22T12:42:51.640096Z","shell.execute_reply":"2021-09-22T12:42:51.649791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-six\"></a>\n## **Final train df**","metadata":{}},{"cell_type":"markdown","source":"**drop columns from df and remain only image id, study id, number of boxes, boxes, labels and image path**","metadata":{}},{"cell_type":"code","source":"train_df.columns","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:51.652028Z","iopub.execute_input":"2021-09-22T12:42:51.652342Z","iopub.status.idle":"2021-09-22T12:42:51.660945Z","shell.execute_reply.started":"2021-09-22T12:42:51.652312Z","shell.execute_reply":"2021-09-22T12:42:51.659936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.drop(columns=['boxes', 'negative', 'typical', 'indeterminate', 'atypical',\n                                  'Rows', 'Columns', 'SOPClassUID','SOPInstanceUID', 'PatientID', 'PatientName', 'PatientSex',\n                                  'PhotometricInterpretation', 'SamplesPerPixel', 'BitsAllocated', 'BitsStored', 'HighBit', \n                                  'PixelRepresentation', 'ImageType', 'Modality', 'BodyPartExamined'])","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:51.662547Z","iopub.execute_input":"2021-09-22T12:42:51.663207Z","iopub.status.idle":"2021-09-22T12:42:51.672933Z","shell.execute_reply.started":"2021-09-22T12:42:51.663149Z","shell.execute_reply":"2021-09-22T12:42:51.671933Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_df.rename(columns={'StudyInstanceUID':'study_id'})","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:51.674644Z","iopub.execute_input":"2021-09-22T12:42:51.675186Z","iopub.status.idle":"2021-09-22T12:42:51.684579Z","shell.execute_reply.started":"2021-09-22T12:42:51.675126Z","shell.execute_reply":"2021-09-22T12:42:51.683629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:51.686284Z","iopub.execute_input":"2021-09-22T12:42:51.686726Z","iopub.status.idle":"2021-09-22T12:42:51.723717Z","shell.execute_reply.started":"2021-09-22T12:42:51.686683Z","shell.execute_reply":"2021-09-22T12:42:51.722748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_csv(os.path.join(output_path, 'train_df.csv'), index=False)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:51.724895Z","iopub.execute_input":"2021-09-22T12:42:51.725185Z","iopub.status.idle":"2021-09-22T12:42:51.868611Z","shell.execute_reply.started":"2021-09-22T12:42:51.725144Z","shell.execute_reply":"2021-09-22T12:42:51.867833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_df)","metadata":{"execution":{"iopub.status.busy":"2021-09-22T12:42:51.869673Z","iopub.execute_input":"2021-09-22T12:42:51.870059Z","iopub.status.idle":"2021-09-22T12:42:51.875196Z","shell.execute_reply.started":"2021-09-22T12:42:51.870033Z","shell.execute_reply":"2021-09-22T12:42:51.874416Z"},"trusted":true},"execution_count":null,"outputs":[]}]}