{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nimport pydicom as dcm\nimport os\nimport cv2\nimport gc\nimport glob\nfrom tqdm import tqdm\nfrom matplotlib.patches import Rectangle\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.applications.mobilenet import preprocess_input","metadata":{"execution":{"iopub.status.busy":"2021-09-05T08:49:06.477351Z","iopub.execute_input":"2021-09-05T08:49:06.477633Z","iopub.status.idle":"2021-09-05T08:49:07.604018Z","shell.execute_reply.started":"2021-09-05T08:49:06.477573Z","shell.execute_reply":"2021-09-05T08:49:07.603002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### EDA","metadata":{}},{"cell_type":"code","source":"labels = pd.read_csv('../input/stage_2_train_labels.csv')\ndetails = pd.read_csv('../input/stage_2_detailed_class_info.csv')","metadata":{"execution":{"iopub.status.busy":"2021-09-05T08:49:07.605811Z","iopub.execute_input":"2021-09-05T08:49:07.60633Z","iopub.status.idle":"2021-09-05T08:49:07.76049Z","shell.execute_reply.started":"2021-09-05T08:49:07.60628Z","shell.execute_reply":"2021-09-05T08:49:07.759595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"details","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:36:50.385781Z","iopub.execute_input":"2021-08-27T10:36:50.386106Z","iopub.status.idle":"2021-08-27T10:36:50.425225Z","shell.execute_reply.started":"2021-08-27T10:36:50.386036Z","shell.execute_reply":"2021-08-27T10:36:50.424178Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# duplicates in details just have the same class so can be safely dropped\ndetails = details.drop_duplicates('patientId').reset_index(drop=True)\nlabels_w_class = labels.merge(details, how='inner', on='patientId')","metadata":{"execution":{"iopub.status.busy":"2021-09-05T08:49:12.309879Z","iopub.execute_input":"2021-09-05T08:49:12.310195Z","iopub.status.idle":"2021-09-05T08:49:12.349937Z","shell.execute_reply.started":"2021-09-05T08:49:12.310136Z","shell.execute_reply":"2021-09-05T08:49:12.349138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_w_class.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:37:17.170572Z","iopub.execute_input":"2021-08-27T10:37:17.170872Z","iopub.status.idle":"2021-08-27T10:37:17.199108Z","shell.execute_reply.started":"2021-08-27T10:37:17.170821Z","shell.execute_reply":"2021-08-27T10:37:17.197796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_w_class[labels_w_class['patientId'] =='c3b05294-29be-46e4-8a51-96fd211e4ca5']","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:37:25.474641Z","iopub.execute_input":"2021-08-27T10:37:25.474958Z","iopub.status.idle":"2021-08-27T10:37:25.506965Z","shell.execute_reply.started":"2021-08-27T10:37:25.474898Z","shell.execute_reply":"2021-08-27T10:37:25.506111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Dealing with missing values","metadata":{}},{"cell_type":"code","source":"labels_w_class.info()\n# No null values in patientId ,Target and Class","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:37:45.782013Z","iopub.execute_input":"2021-08-27T10:37:45.78233Z","iopub.status.idle":"2021-08-27T10:37:45.802191Z","shell.execute_reply.started":"2021-08-27T10:37:45.782274Z","shell.execute_reply":"2021-08-27T10:37:45.801092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# null values in x, y, width, height indicates that there is no pneumonia. Replacing null with 0\nlabels_w_class.fillna(0, inplace=True)\nlabels_w_class.info()","metadata":{"execution":{"iopub.status.busy":"2021-09-05T08:49:14.953337Z","iopub.execute_input":"2021-09-05T08:49:14.953637Z","iopub.status.idle":"2021-09-05T08:49:14.993435Z","shell.execute_reply.started":"2021-09-05T08:49:14.953582Z","shell.execute_reply":"2021-09-05T08:49:14.992674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualization of different classes","metadata":{}},{"cell_type":"code","source":"labels_w_class['class'].unique()\n# Three categories of classes are present","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:38:12.935214Z","iopub.execute_input":"2021-08-27T10:38:12.935559Z","iopub.status.idle":"2021-08-27T10:38:12.943984Z","shell.execute_reply.started":"2021-08-27T10:38:12.935504Z","shell.execute_reply":"2021-08-27T10:38:12.942975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.unique(labels_w_class[\"Target\"])\n# Two values of targets are present","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:38:20.081911Z","iopub.execute_input":"2021-08-27T10:38:20.082222Z","iopub.status.idle":"2021-08-27T10:38:20.088614Z","shell.execute_reply.started":"2021-08-27T10:38:20.082158Z","shell.execute_reply":"2021-08-27T10:38:20.087605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(labels_w_class[\"Target\"])\n# The number of '0' category is almost double than that of '1' category","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:38:27.546914Z","iopub.execute_input":"2021-08-27T10:38:27.547249Z","iopub.status.idle":"2021-08-27T10:38:27.732328Z","shell.execute_reply.started":"2021-08-27T10:38:27.547181Z","shell.execute_reply":"2021-08-27T10:38:27.731403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(labels_w_class[\"class\"])","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:38:36.570214Z","iopub.execute_input":"2021-08-27T10:38:36.570556Z","iopub.status.idle":"2021-08-27T10:38:36.786373Z","shell.execute_reply.started":"2021-08-27T10:38:36.5705Z","shell.execute_reply":"2021-08-27T10:38:36.785557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dicom_file_path = os.path.join(\"../input/stage_2_train_images/0004cfab-14fd-4e49-80ba-63a80b6bddd6.dcm\")\nfile = dcm.read_file(dicom_file_path)\nfile\n# The given images contains lot of metadata information","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:38:51.316772Z","iopub.execute_input":"2021-08-27T10:38:51.317095Z","iopub.status.idle":"2021-08-27T10:38:51.338845Z","shell.execute_reply.started":"2021-08-27T10:38:51.317038Z","shell.execute_reply":"2021-08-27T10:38:51.338156Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"column_list = [\"Patient ID\", \"Patient Sex\", \"Patient's Age\", \"View Position\",\"Image Size\"]\nfile_meta_Data = pd.DataFrame(columns=column_list)","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:39:02.357149Z","iopub.execute_input":"2021-08-27T10:39:02.357521Z","iopub.status.idle":"2021-08-27T10:39:02.369924Z","shell.execute_reply.started":"2021-08-27T10:39:02.357464Z","shell.execute_reply":"2021-08-27T10:39:02.369123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_meta_data_to_df(df, loc, from_list):\n    data = []\n    for filename in from_list:\n            imagePath = loc+filename\n            data_row_img_data = dcm.read_file(imagePath)\n            values = []\n            values.append(data_row_img_data.PatientID.strip())\n            values.append(data_row_img_data.PatientSex)\n            values.append(data_row_img_data.PatientAge)\n            values.append(data_row_img_data.ViewPosition)\n            values.append(f\"{data_row_img_data.Rows}x{data_row_img_data.Columns}\")\n            zipped_val = dict(zip(column_list, values))\n            df = df.append(zipped_val, True)\n    return df","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:39:09.72927Z","iopub.execute_input":"2021-08-27T10:39:09.729588Z","iopub.status.idle":"2021-08-27T10:39:09.738324Z","shell.execute_reply.started":"2021-08-27T10:39:09.729535Z","shell.execute_reply":"2021-08-27T10:39:09.736501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Images Example\ntrain_images_dir = '../input/stage_2_train_images/'\ntrain_images = [f for f in os.listdir(train_images_dir) if os.path.isfile(os.path.join(train_images_dir, f))]\ntest_images_dir = '../input/stage_2_test_images/'\ntest_images = [f for f in os.listdir(test_images_dir) if os.path.isfile(os.path.join(test_images_dir, f))]\nprint('5 Training images', train_images[:5]) # Print the first 5","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:39:26.773788Z","iopub.execute_input":"2021-08-27T10:39:26.774094Z","iopub.status.idle":"2021-08-27T10:40:21.264016Z","shell.execute_reply.started":"2021-08-27T10:39:26.774038Z","shell.execute_reply":"2021-08-27T10:40:21.263179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(train_images))\nprint(len(test_images))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_meta_Data = add_meta_data_to_df(file_meta_Data, \"../input/stage_2_train_images/\", train_images)","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:40:21.265518Z","iopub.execute_input":"2021-08-27T10:40:21.265985Z","iopub.status.idle":"2021-08-27T10:45:00.677531Z","shell.execute_reply.started":"2021-08-27T10:40:21.265931Z","shell.execute_reply":"2021-08-27T10:45:00.676515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_meta_Data.info()","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:46:24.742959Z","iopub.execute_input":"2021-08-27T10:46:24.743308Z","iopub.status.idle":"2021-08-27T10:46:24.768081Z","shell.execute_reply.started":"2021-08-27T10:46:24.743224Z","shell.execute_reply":"2021-08-27T10:46:24.767067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_df = labels_w_class.copy()\ntemp_df['ratio'] = (temp_df[\"width\"] * temp_df[\"height\"])/(1024*1024)","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:45:00.716447Z","iopub.execute_input":"2021-08-27T10:45:00.716922Z","iopub.status.idle":"2021-08-27T10:45:00.782254Z","shell.execute_reply.started":"2021-08-27T10:45:00.716866Z","shell.execute_reply":"2021-08-27T10:45:00.781292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_df['ratio'].unique()","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:45:00.783376Z","iopub.execute_input":"2021-08-27T10:45:00.783626Z","iopub.status.idle":"2021-08-27T10:45:00.795987Z","shell.execute_reply.started":"2021-08-27T10:45:00.783581Z","shell.execute_reply":"2021-08-27T10:45:00.795196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_info_train_labels_merge_metadata = labels_w_class.merge(file_meta_Data,how = \"left\",right_on = 'Patient ID',left_on = 'patientId') \nclass_info_train_labels_merge_metadata['xc'] = class_info_train_labels_merge_metadata['x'] + class_info_train_labels_merge_metadata['width'] / 2\nclass_info_train_labels_merge_metadata['yc'] = class_info_train_labels_merge_metadata['y'] + class_info_train_labels_merge_metadata['height'] / 2","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:45:13.898992Z","iopub.execute_input":"2021-08-27T10:45:13.899335Z","iopub.status.idle":"2021-08-27T10:45:13.936812Z","shell.execute_reply.started":"2021-08-27T10:45:13.899272Z","shell.execute_reply":"2021-08-27T10:45:13.935941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_info_train_labels_merge_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:45:15.585835Z","iopub.execute_input":"2021-08-27T10:45:15.586157Z","iopub.status.idle":"2021-08-27T10:45:15.623353Z","shell.execute_reply.started":"2021-08-27T10:45:15.586102Z","shell.execute_reply":"2021-08-27T10:45:15.622528Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_window(data,color_point, color_window,text):\n    fig, ax = plt.subplots(1,1,figsize=(4,4))\n    plt.title(\"Centers of Lung Opacity rectangles over rectangles\\n{}\".format(text))\n    data.plot.scatter(x='xc', y='yc', xlim=(0,1024), ylim=(0,1024), ax=ax, alpha=0.8, marker=\".\", color=color_point)\n    for i, crt_sample in data.iterrows():\n        ax.add_patch(Rectangle(xy=(crt_sample['x'], crt_sample['y']),\n            width=crt_sample['width'],height=crt_sample['height'],alpha=3.5e-3, color=color_window))\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:45:19.313689Z","iopub.execute_input":"2021-08-27T10:45:19.31425Z","iopub.status.idle":"2021-08-27T10:45:19.323097Z","shell.execute_reply.started":"2021-08-27T10:45:19.313953Z","shell.execute_reply":"2021-08-27T10:45:19.3222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classify = (class_info_train_labels_merge_metadata['View Position']=='AP') \n\nplot_window(class_info_train_labels_merge_metadata[ classify ],'green', 'yellow', 'Patient View Position: AP')","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:45:21.430876Z","iopub.execute_input":"2021-08-27T10:45:21.431169Z","iopub.status.idle":"2021-08-27T10:45:42.250117Z","shell.execute_reply.started":"2021-08-27T10:45:21.431114Z","shell.execute_reply":"2021-08-27T10:45:42.249218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classify = (class_info_train_labels_merge_metadata['View Position']=='PA') \n\nplot_window(class_info_train_labels_merge_metadata[ classify ],'blue', 'red', 'Patient View Position: PA')\n# Distribution is slightly different for view position  PA","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:45:42.251781Z","iopub.execute_input":"2021-08-27T10:45:42.252063Z","iopub.status.idle":"2021-08-27T10:45:59.830317Z","shell.execute_reply.started":"2021-08-27T10:45:42.252018Z","shell.execute_reply":"2021-08-27T10:45:59.829358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.distplot(file_meta_Data[\"Patient's Age\"].astype(int))","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:46:58.481796Z","iopub.execute_input":"2021-08-27T10:46:58.482374Z","iopub.status.idle":"2021-08-27T10:46:58.876816Z","shell.execute_reply.started":"2021-08-27T10:46:58.482094Z","shell.execute_reply":"2021-08-27T10:46:58.874268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_images(data):\n    img_data = list(data.T.to_dict().values())\n    f, ax = plt.subplots(3,3, figsize=(16,18))\n    for i,data_row in enumerate(img_data):\n        patientImage = data_row['patientId']+'.dcm'\n        imagePath = f\"../input/stage_2_train_images/{patientImage}\"\n        data_row_img_data = dcm.read_file(imagePath)\n        #modality = data_row_img_data.Modality\n        age = data_row_img_data.PatientAge\n        sex = data_row_img_data.PatientSex\n        data_row_img = dcm.dcmread(imagePath)\n        ax[i//3, i%3].imshow(data_row_img.pixel_array, cmap=plt.cm.bone) \n        ax[i//3, i%3].axis('off')\n        ax[i//3, i%3].set_title('ID: {} \\nAge: {} Sex: {} Target: {}\\nWindow: {}:{}:{}:{}'.format(\n                data_row['patientId']\n                , age, sex, data_row['Target'], \n                data_row['x'],data_row['y'],data_row['width'],data_row['height']))\n    plt.show()\n\nshow_images(labels_w_class[labels_w_class['Target']==1].sample(9))","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:47:10.396217Z","iopub.execute_input":"2021-08-27T10:47:10.396545Z","iopub.status.idle":"2021-08-27T10:47:12.787441Z","shell.execute_reply.started":"2021-08-27T10:47:10.396493Z","shell.execute_reply":"2021-08-27T10:47:12.786475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_dicom_images_with_boxes(data):\n    img_data = list(data.T.to_dict().values())\n    f, ax = plt.subplots(3,3, figsize=(14,16))\n    for i,data_row in enumerate(img_data):\n        patientImage = data_row['patientId']+'.dcm'\n        imagePath = f\"../input/stage_2_train_images/{patientImage}\"\n        data_row_img_data = dcm.read_file(imagePath)\n        #modality = data_row_img_data.Modality\n        age = data_row_img_data.PatientAge\n        sex = data_row_img_data.PatientSex\n        data_row_img = dcm.dcmread(imagePath)\n        ax[i//3, i%3].imshow(data_row_img.pixel_array, cmap=plt.cm.bone) \n        ax[i//3, i%3].axis('off')\n        ax[i//3, i%3].set_title('ID: {}\\n Age: {} Sex: {} Target: {}'.format(\n                data_row['patientId'], age, sex, data_row['Target']))\n        rows = labels_w_class[labels_w_class['patientId']==data_row['patientId']]\n        box_data = list(rows.T.to_dict().values())\n        for j, row in enumerate(box_data):\n            ax[i//3, i%3].add_patch(Rectangle(xy=(row['x'], row['y']),\n                        width=row['width'],height=row['height'], \n                        color=\"yellow\",alpha = 0.1))   \n    plt.show()\n    \nshow_dicom_images_with_boxes(labels_w_class[labels_w_class['Target']==1].sample(9))","metadata":{"execution":{"iopub.status.busy":"2021-08-27T10:47:21.518791Z","iopub.execute_input":"2021-08-27T10:47:21.519086Z","iopub.status.idle":"2021-08-27T10:47:23.028841Z","shell.execute_reply.started":"2021-08-27T10:47:21.51903Z","shell.execute_reply":"2021-08-27T10:47:23.02809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Unet-Resnet","metadata":{}},{"cell_type":"code","source":"import os\nimport csv\nimport random\nimport pydicom\nimport numpy as np\nimport pandas as pd\nfrom skimage import measure\nfrom skimage.transform import resize\n\nimport tensorflow as tf\nfrom tensorflow import keras\n\nfrom matplotlib import pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-09-05T08:49:28.690875Z","iopub.execute_input":"2021-09-05T08:49:28.691183Z","iopub.status.idle":"2021-09-05T08:49:28.964129Z","shell.execute_reply.started":"2021-09-05T08:49:28.691133Z","shell.execute_reply":"2021-09-05T08:49:28.963352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# empty dictionary\npneumonia_locations = {}\n# load table\nwith open(os.path.join('../input/stage_2_train_labels.csv'), mode='r') as infile:\n    # open reader\n    reader = csv.reader(infile)\n    # skip header\n    next(reader, None)\n    # loop through rows\n    for rows in reader:\n        # retrieve information\n        filename = rows[0]\n        location = rows[1:5]\n        pneumonia = rows[5]\n        # if row contains pneumonia add label to dictionary\n        # which contains a list of pneumonia locations per filename\n        if pneumonia == '1':\n            # convert string to float to int\n            location = [int(float(i)) for i in location]\n            # save pneumonia location in dictionary\n            if filename in pneumonia_locations:\n                pneumonia_locations[filename].append(location)\n            else:\n                pneumonia_locations[filename] = [location]","metadata":{"_uuid":"e08496a85ef9b0823595c3745d2677c6e84b6a3a","execution":{"iopub.status.busy":"2021-09-05T08:49:30.12741Z","iopub.execute_input":"2021-09-05T08:49:30.127728Z","iopub.status.idle":"2021-09-05T08:49:30.226808Z","shell.execute_reply.started":"2021-09-05T08:49:30.127654Z","shell.execute_reply":"2021-09-05T08:49:30.225914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load and shuffle filenames\nfolder = '../input/stage_2_train_images'\nfilenames = os.listdir(folder)\nrandom.shuffle(filenames)\n# split into train and validation filenames\nn_valid_samples = 2000\nn_train_samples = 17000\ntrain_filenames = filenames[n_train_samples:]\nvalid_filenames = filenames[:n_valid_samples]\nprint('n train samples', len(train_filenames))\nprint('n valid samples', len(valid_filenames))\nn_train_samples = len(filenames) - n_valid_samples","metadata":{"_uuid":"ccd0b0d52cafd125558ed5560a9cc8fa15760bc5","execution":{"iopub.status.busy":"2021-09-05T08:49:32.473263Z","iopub.execute_input":"2021-09-05T08:49:32.473544Z","iopub.status.idle":"2021-09-05T08:49:32.973941Z","shell.execute_reply.started":"2021-09-05T08:49:32.473496Z","shell.execute_reply":"2021-09-05T08:49:32.973107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_filenames","metadata":{"execution":{"iopub.status.busy":"2021-08-24T12:48:51.561156Z","iopub.execute_input":"2021-08-24T12:48:51.561438Z","iopub.status.idle":"2021-08-24T12:48:51.577245Z","shell.execute_reply.started":"2021-08-24T12:48:51.561379Z","shell.execute_reply":"2021-08-24T12:48:51.576506Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ### Data generator\n\nThe dataset is too large to fit into memory, so we need to create a generator that loads data on the fly.\n\n* The generator takes in some filenames, batch_size and other parameters.\n\n* The generator outputs a random batch of numpy images and numpy masks.\n    ","metadata":{"_uuid":"276b80f59fa9acdfd9307d74cee5f705cd6aa5b6","trusted":true}},{"cell_type":"code","source":"class generator(keras.utils.Sequence):\n    \n    def __init__(self, folder, filenames, pneumonia_locations=None, batch_size=32, image_size=224, shuffle=True, augment=False, predict=False):\n        self.folder = folder\n        self.filenames = filenames\n        self.pneumonia_locations = pneumonia_locations\n        self.batch_size = batch_size\n        self.image_size = image_size\n        self.shuffle = shuffle\n        self.augment = augment\n        self.predict = predict\n        self.on_epoch_end()\n        \n    def __load__(self, filename):\n        # load dicom file as numpy array\n        img = pydicom.dcmread(os.path.join(self.folder, filename)).pixel_array\n        # create empty mask\n        msk = np.zeros(img.shape)\n        # get filename without extension\n        filename = filename.split('.')[0]\n        # if image contains pneumonia\n        if filename in pneumonia_locations:\n            # loop through pneumonia\n            for location in pneumonia_locations[filename]:\n                # add 1's at the location of the pneumonia\n                x, y, w, h = location\n                msk[y:y+h, x:x+w] = 1\n        # if augment then horizontal flip half the time\n        if self.augment and random.random() > 0.5:\n            img = np.fliplr(img)\n            msk = np.fliplr(msk)\n        # resize both image and mask\n        img = resize(img, (self.image_size, self.image_size), mode='reflect')\n        msk = resize(msk, (self.image_size, self.image_size), mode='reflect') > 0.5\n        # add trailing channel dimension\n        img = np.expand_dims(img, -1)\n        msk = np.expand_dims(msk, -1)\n        return img, msk\n    \n    def __loadpredict__(self, filename):\n        # load dicom file as numpy array\n        img = pydicom.dcmread(os.path.join(self.folder, filename)).pixel_array\n        # resize image\n        img = resize(img, (self.image_size, self.image_size), mode='reflect')\n        # add trailing channel dimension\n        img = np.expand_dims(img, -1)\n        return img\n        \n    def __getitem__(self, index):\n        # select batch\n        filenames = self.filenames[index*self.batch_size:(index+1)*self.batch_size]\n        # predict mode: return images and filenames\n        if self.predict:\n            # load files\n            imgs = [self.__loadpredict__(filename) for filename in filenames]\n            # create numpy batch\n            imgs = np.array(imgs)\n            return imgs, filenames\n        # train mode: return images and masks\n        else:\n            # load files\n            items = [self.__load__(filename) for filename in filenames]\n            # unzip images and masks\n            imgs, msks = zip(*items)\n            # create numpy batch\n            imgs = np.array(imgs)\n            msks = np.array(msks)\n            return imgs, msks\n        \n    def on_epoch_end(self):\n        if self.shuffle:\n            random.shuffle(self.filenames)\n        \n    def __len__(self):\n        if self.predict:\n            # return everything\n            return int(np.ceil(len(self.filenames) / self.batch_size))\n        else:\n            # return full batches only\n            return int(len(self.filenames) / self.batch_size)","metadata":{"_uuid":"86b3f780a03cddda78c6adfde461d6ff8dad5672","execution":{"iopub.status.busy":"2021-09-05T08:49:35.094203Z","iopub.execute_input":"2021-09-05T08:49:35.094534Z","iopub.status.idle":"2021-09-05T08:49:35.117461Z","shell.execute_reply.started":"2021-09-05T08:49:35.094482Z","shell.execute_reply":"2021-09-05T08:49:35.11644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 16\nIMAGE_SIZE = 224\n\ndef create_downsample(channels, inputs):\n    x = keras.layers.BatchNormalization(momentum=0.9)(inputs)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(channels, 1, padding='same', use_bias=False)(x)\n    x = keras.layers.MaxPool2D(2)(x)\n    return x\n\ndef create_resblock(channels, inputs):\n    x = keras.layers.BatchNormalization(momentum=0.9)(inputs)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(channels, 3, padding='same', use_bias=False)(x)\n    x = keras.layers.BatchNormalization(momentum=0.9)(x)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(channels, 3, padding='same', use_bias=False)(x)\n    return keras.layers.add([x, inputs])\n\ndef create_network(input_size, channels, n_blocks=2, depth=4):\n    # input\n    inputs = keras.Input(shape=(input_size, input_size, 1))\n    x = keras.layers.Conv2D(channels, 3, padding='same', use_bias=False)(inputs)\n    # residual blocks\n    for d in range(depth):\n        channels = channels * 2\n        x = create_downsample(channels, x)\n        for b in range(n_blocks):\n            x = create_resblock(channels, x)\n    # output\n    #x = keras.layers.BatchNormalization(momentum=0.9)(x)\n    #x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(256, 1, activation=None)(x)\n    x = keras.layers.BatchNormalization(momentum=0.9)(x)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2DTranspose(128, (8,8), (4,4), padding=\"same\", activation=None)(x)\n    x = keras.layers.BatchNormalization(momentum=0.9)(x)\n    x = keras.layers.LeakyReLU(0)(x)\n    x = keras.layers.Conv2D(1, 1, activation='sigmoid')(x)\n    outputs = keras.layers.UpSampling2D(2**(depth-2))(x)\n    model = keras.Model(inputs=inputs, outputs=outputs)\n    return model","metadata":{"_uuid":"9fc2b108689637a6037b48ebab3f7659b8704bf9","execution":{"iopub.status.busy":"2021-09-05T08:49:38.184733Z","iopub.execute_input":"2021-09-05T08:49:38.185037Z","iopub.status.idle":"2021-09-05T08:49:38.203573Z","shell.execute_reply.started":"2021-09-05T08:49:38.184989Z","shell.execute_reply":"2021-09-05T08:49:38.202776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define iou or jaccard loss function\ndef iou_loss(y_true, y_pred):\n    y_true = tf.reshape(y_true, [-1])\n    y_pred = tf.reshape(y_pred, [-1])\n    intersection = tf.reduce_sum(y_true * y_pred)\n    score = (intersection + 1.) / (tf.reduce_sum(y_true) + tf.reduce_sum(y_pred) - intersection + 1.)\n    return 1 - score\n\n# combine bce loss and iou loss\ndef iou_bce_loss(y_true, y_pred):\n    return 0.5 * keras.losses.binary_crossentropy(y_true, y_pred) + 0.5 * iou_loss(y_true, y_pred)\n\n# mean iou as a metric\ndef mean_iou(y_true, y_pred):\n    y_pred = tf.round(y_pred)\n    intersect = tf.reduce_sum(y_true * y_pred, axis=[1, 2, 3])\n    union = tf.reduce_sum(y_true, axis=[1, 2, 3]) + tf.reduce_sum(y_pred, axis=[1, 2, 3])\n    smooth = tf.ones(tf.shape(intersect))\n    return tf.reduce_mean((intersect + smooth) / (union - intersect + smooth))\n\n# create network and compiler\nmodel = create_network(input_size=IMAGE_SIZE, channels=32, n_blocks=2, depth=4)\nmodel.compile(optimizer='adam',\n              loss=iou_bce_loss,\n              metrics=['accuracy', mean_iou])\n\n# cosine learning rate annealing\ndef cosine_annealing(x):\n    lr = 0.001\n    epochs = 20\n    return lr*(np.cos(np.pi*x/epochs)+1.)/2\nlearning_rate = tf.keras.callbacks.LearningRateScheduler(cosine_annealing)\n\n# create train and validation generators\nfolder = '../input/stage_2_train_images'\ntrain_gen = generator(folder, train_filenames, pneumonia_locations, batch_size=BATCH_SIZE, image_size=IMAGE_SIZE, shuffle=True, augment=True, predict=False)\nvalid_gen = generator(folder, valid_filenames, pneumonia_locations, batch_size=BATCH_SIZE, image_size=IMAGE_SIZE, shuffle=False, predict=False)\n\n#print(model.summary())","metadata":{"_uuid":"4369be30f61440eb6858d57829fa541c4ee893bf","execution":{"iopub.status.busy":"2021-09-05T08:49:40.209745Z","iopub.execute_input":"2021-09-05T08:49:40.210045Z","iopub.status.idle":"2021-09-05T08:49:41.917109Z","shell.execute_reply.started":"2021-09-05T08:49:40.209995Z","shell.execute_reply":"2021-09-05T08:49:41.915818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit_generator(train_gen, validation_data=valid_gen,  epochs=16, shuffle=True)","metadata":{"_uuid":"f6bb70654592bc3b47f50f09982d42f4c6971463","execution":{"iopub.status.busy":"2021-09-05T08:49:46.52019Z","iopub.execute_input":"2021-09-05T08:49:46.520464Z","iopub.status.idle":"2021-09-05T10:12:27.82257Z","shell.execute_reply.started":"2021-09-05T08:49:46.520416Z","shell.execute_reply":"2021-09-05T10:12:27.821633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install visualkeras","metadata":{"execution":{"iopub.status.busy":"2021-09-05T10:14:45.762193Z","iopub.execute_input":"2021-09-05T10:14:45.762545Z","iopub.status.idle":"2021-09-05T10:15:16.516344Z","shell.execute_reply.started":"2021-09-05T10:14:45.762498Z","shell.execute_reply":"2021-09-05T10:15:16.515395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import visualkeras\n\nmodel = model\n\n#visualkeras.layered_view(model).show() # display using your system viewer\n#visualkeras.layered_view(model, to_file='output.png') # write to disk\nvisualkeras.layered_view(model, to_file='output.png',legend=True) # write and show","metadata":{"execution":{"iopub.status.busy":"2021-09-05T10:19:22.67627Z","iopub.execute_input":"2021-09-05T10:19:22.676612Z","iopub.status.idle":"2021-09-05T10:19:23.240015Z","shell.execute_reply.started":"2021-09-05T10:19:22.676549Z","shell.execute_reply":"2021-09-05T10:19:23.238998Z"},"trusted":true},"execution_count":null,"outputs":[]}]}