{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from functools import partial\nfrom collections import defaultdict\nimport pydicom\nimport os\nimport glob\nimport numpy as np\nimport pandas as pd\nfrom joblib import Parallel, delayed\nfrom lightgbm import LGBMClassifier\nfrom sklearn.model_selection import KFold\nfrom sklearn.metrics import roc_auc_score\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport seaborn as sns\nfrom tqdm import tqdm\nsns.set_style('whitegrid')\n%matplotlib inline\n\nnp.warnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:35:26.731894Z","iopub.execute_input":"2021-07-05T13:35:26.732261Z","iopub.status.idle":"2021-07-05T13:35:29.986091Z","shell.execute_reply.started":"2021-07-05T13:35:26.732187Z","shell.execute_reply":"2021-07-05T13:35:29.985160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_train_labels.csv')\ndetails = pd.read_csv('../input/rsna-pneumonia-detection-challenge/stage_2_detailed_class_info.csv')\n\n# duplicates in details just have the same class so can be safely dropped\ndetails = details.drop_duplicates('patientId').reset_index(drop=True)\nlabels_w_class = labels.merge(details, how='inner', on='patientId')","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:35:29.987507Z","iopub.execute_input":"2021-07-05T13:35:29.987849Z","iopub.status.idle":"2021-07-05T13:35:30.129131Z","shell.execute_reply.started":"2021-07-05T13:35:29.987816Z","shell.execute_reply":"2021-07-05T13:35:30.128348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_w_class","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:35:30.130930Z","iopub.execute_input":"2021-07-05T13:35:30.131267Z","iopub.status.idle":"2021-07-05T13:35:30.155679Z","shell.execute_reply.started":"2021-07-05T13:35:30.131233Z","shell.execute_reply":"2021-07-05T13:35:30.154906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get lists of all train/test dicom filepaths\ntrain_dcm_fps = glob.glob('../input/rsna-pneumonia-detection-challenge/stage_2_train_images/*.dcm')\ntest_dcm_fps = glob.glob('../input/rsna-pneumonia-detection-challenge/stage_2_test_images/*.dcm')\n\ntrain_dcm_fps = train_dcm_fps[:11000]\ntest_dcm_fps = test_dcm_fps[:3000]\n\n# read each file into a list (using stop_before_pixels to avoid reading the image for speed and memory savings)\ntrain_dcms = [pydicom.read_file(x, stop_before_pixels=True) for x in tqdm(train_dcm_fps)]\ntest_dcms = [pydicom.read_file(x, stop_before_pixels=True) for x in tqdm(test_dcm_fps)]","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:35:30.157331Z","iopub.execute_input":"2021-07-05T13:35:30.157650Z","iopub.status.idle":"2021-07-05T13:37:14.922367Z","shell.execute_reply.started":"2021-07-05T13:35:30.157616Z","shell.execute_reply":"2021-07-05T13:37:14.921079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_dcm_fps)","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:14.923647Z","iopub.execute_input":"2021-07-05T13:37:14.923991Z","iopub.status.idle":"2021-07-05T13:37:14.930370Z","shell.execute_reply.started":"2021-07-05T13:37:14.923956Z","shell.execute_reply":"2021-07-05T13:37:14.929394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dcm_fps[0]","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:14.931635Z","iopub.execute_input":"2021-07-05T13:37:14.932314Z","iopub.status.idle":"2021-07-05T13:37:14.941732Z","shell.execute_reply.started":"2021-07-05T13:37:14.932275Z","shell.execute_reply":"2021-07-05T13:37:14.940924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dcms[1]","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:14.943148Z","iopub.execute_input":"2021-07-05T13:37:14.943578Z","iopub.status.idle":"2021-07-05T13:37:14.954140Z","shell.execute_reply.started":"2021-07-05T13:37:14.943544Z","shell.execute_reply":"2021-07-05T13:37:14.953390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def parse_dcm_metadata(dcm):\n    unpacked_data = {}\n    group_elem_to_keywords = {}\n    # iterating here to force conversion from lazy RawDataElement to DataElement\n    for d in dcm:\n        pass\n    # keys are pydicom.tag.BaseTag, values are pydicom.dataelem.DataElement\n    for tag, elem in dcm.items():\n        tag_group = tag.group\n        tag_elem = tag.elem\n        keyword = elem.keyword\n        group_elem_to_keywords[(tag_group, tag_elem)] = keyword\n        value = elem.value\n        unpacked_data[keyword] = value\n    return unpacked_data, group_elem_to_keywords\n\ntrain_meta_dicts, tag_to_keyword_train = zip(*[parse_dcm_metadata(x) for x in tqdm(train_dcms)])\ntest_meta_dicts, tag_to_keyword_test = zip(*[parse_dcm_metadata(x) for x in tqdm(test_dcms)])","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:14.957764Z","iopub.execute_input":"2021-07-05T13:37:14.958060Z","iopub.status.idle":"2021-07-05T13:37:25.186728Z","shell.execute_reply.started":"2021-07-05T13:37:14.958031Z","shell.execute_reply":"2021-07-05T13:37:25.185779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tag_to_keyword_train[0]\ntrain_meta_dicts[0]","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:25.188689Z","iopub.execute_input":"2021-07-05T13:37:25.189057Z","iopub.status.idle":"2021-07-05T13:37:25.195688Z","shell.execute_reply.started":"2021-07-05T13:37:25.189017Z","shell.execute_reply":"2021-07-05T13:37:25.194790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# join all the dicts\nunified_tag_to_key_train = {k:v for dict_ in tag_to_keyword_train for k,v in dict_.items()}\nunified_tag_to_key_test = {k:v for dict_ in tag_to_keyword_test for k,v in dict_.items()}\n\n# quick check to make sure there are no different keys between test/train\nassert len(set(unified_tag_to_key_test.keys()).symmetric_difference(set(unified_tag_to_key_train.keys()))) == 0\n\ntag_to_key = {**unified_tag_to_key_test, **unified_tag_to_key_train}\ntag_to_key","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:25.197091Z","iopub.execute_input":"2021-07-05T13:37:25.197464Z","iopub.status.idle":"2021-07-05T13:37:25.262343Z","shell.execute_reply.started":"2021-07-05T13:37:25.197427Z","shell.execute_reply":"2021-07-05T13:37:25.261626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using from_records here since some values in the dicts will be iterables and some are constants\ntrain_df = pd.DataFrame.from_dict(data=train_meta_dicts)\ntest_df = pd.DataFrame.from_dict(data=test_meta_dicts)\ntrain_df['dataset'] = 'train'\ntest_df['dataset'] = 'test'\n#df = pd.concat([train_df, test_df])\ndf = train_df\ndf2 = test_df","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:25.263388Z","iopub.execute_input":"2021-07-05T13:37:25.263717Z","iopub.status.idle":"2021-07-05T13:37:25.449113Z","shell.execute_reply.started":"2021-07-05T13:37:25.263668Z","shell.execute_reply":"2021-07-05T13:37:25.448283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:25.451520Z","iopub.execute_input":"2021-07-05T13:37:25.452081Z","iopub.status.idle":"2021-07-05T13:37:25.509417Z","shell.execute_reply.started":"2021-07-05T13:37:25.452044Z","shell.execute_reply":"2021-07-05T13:37:25.508437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:25.510949Z","iopub.execute_input":"2021-07-05T13:37:25.511330Z","iopub.status.idle":"2021-07-05T13:37:25.557470Z","shell.execute_reply.started":"2021-07-05T13:37:25.511290Z","shell.execute_reply":"2021-07-05T13:37:25.556651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#[1,0] for PA and [0,1] for AP\n# y=df['SeriesDescription']=='view: PA'\n# number_of_images = len(y)\n# train_Y = np.zeros((number_of_images,2))\n# for i in range(0,number_of_images):\n#     if(y[i] == True):\n#         train_Y[i] = [1,0]\n#     else:\n#         train_Y[i] = [0,1]\n        \n# train_Y\n","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:25.558632Z","iopub.execute_input":"2021-07-05T13:37:25.559121Z","iopub.status.idle":"2021-07-05T13:37:25.562746Z","shell.execute_reply.started":"2021-07-05T13:37:25.559083Z","shell.execute_reply":"2021-07-05T13:37:25.561883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=df['SeriesDescription']=='view: PA'\nnumber_of_images = len(y)\ntrain_Y = np.zeros(number_of_images)\nfor i in range(0,number_of_images):\n    if(y[i] == True):\n        train_Y[i] = 1\n        \ntrain_Y","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:25.564281Z","iopub.execute_input":"2021-07-05T13:37:25.564635Z","iopub.status.idle":"2021-07-05T13:37:25.648164Z","shell.execute_reply.started":"2021-07-05T13:37:25.564600Z","shell.execute_reply":"2021-07-05T13:37:25.647228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_Y.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:25.649276Z","iopub.execute_input":"2021-07-05T13:37:25.649753Z","iopub.status.idle":"2021-07-05T13:37:25.658392Z","shell.execute_reply.started":"2021-07-05T13:37:25.649703Z","shell.execute_reply":"2021-07-05T13:37:25.657453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_test = df2['SeriesDescription']=='view: PA'\n# number_of_images = len(y_test)\n# test_Y = np.zeros((number_of_images,2))\n# for i in range(0,number_of_images):\n#     if(y_test[i] == True):\n#         test_Y[i] = [1,0]\n#     else:\n#         test_Y[i] = [0,1]\n        \n# test_Y","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:25.659826Z","iopub.execute_input":"2021-07-05T13:37:25.660119Z","iopub.status.idle":"2021-07-05T13:37:25.665670Z","shell.execute_reply.started":"2021-07-05T13:37:25.660092Z","shell.execute_reply":"2021-07-05T13:37:25.664775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = df2['SeriesDescription']=='view: PA'\nnumber_of_images = len(y_test)\ntest_Y = np.zeros(number_of_images)\nfor i in range(0,number_of_images):\n    if(y_test[i] == True):\n        test_Y[i] = 1\n\ntest_Y       ","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:25.667467Z","iopub.execute_input":"2021-07-05T13:37:25.668177Z","iopub.status.idle":"2021-07-05T13:37:25.705152Z","shell.execute_reply.started":"2021-07-05T13:37:25.668138Z","shell.execute_reply":"2021-07-05T13:37:25.704418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\ntrain_X=[]\nfor x in tqdm(train_dcm_fps):\n    img = pydicom.read_file(x).pixel_array\n    img = cv2.resize(img, (128, 128))\n    img = img/255\n    train_X.append(img)","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:37:25.706280Z","iopub.execute_input":"2021-07-05T13:37:25.706601Z","iopub.status.idle":"2021-07-05T13:38:46.105584Z","shell.execute_reply.started":"2021-07-05T13:37:25.706567Z","shell.execute_reply":"2021-07-05T13:38:46.104742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = np.array(train_X)","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:38:46.106790Z","iopub.execute_input":"2021-07-05T13:38:46.107105Z","iopub.status.idle":"2021-07-05T13:38:46.579176Z","shell.execute_reply.started":"2021-07-05T13:38:46.107078Z","shell.execute_reply":"2021-07-05T13:38:46.578058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X[0]","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:38:46.580651Z","iopub.execute_input":"2021-07-05T13:38:46.581218Z","iopub.status.idle":"2021-07-05T13:38:46.587991Z","shell.execute_reply.started":"2021-07-05T13:38:46.581180Z","shell.execute_reply":"2021-07-05T13:38:46.587168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(train_X[170],cmap = 'gray')","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:38:46.589662Z","iopub.execute_input":"2021-07-05T13:38:46.590281Z","iopub.status.idle":"2021-07-05T13:38:46.895313Z","shell.execute_reply.started":"2021-07-05T13:38:46.590241Z","shell.execute_reply":"2021-07-05T13:38:46.894587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X_rgb = np.repeat(train_X[..., np.newaxis], 3, -1)\nprint(train_X_rgb.shape)  ","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:38:46.898649Z","iopub.execute_input":"2021-07-05T13:38:46.899199Z","iopub.status.idle":"2021-07-05T13:38:49.619961Z","shell.execute_reply.started":"2021-07-05T13:38:46.899163Z","shell.execute_reply":"2021-07-05T13:38:49.619089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_X=[]\nfor x in tqdm(test_dcm_fps):\n    img_test = pydicom.read_file(x).pixel_array\n    img_test = cv2.resize(img_test, (128, 128))\n    img_test = img_test/255\n    test_X.append(img_test)","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:38:49.621755Z","iopub.execute_input":"2021-07-05T13:38:49.622197Z","iopub.status.idle":"2021-07-05T13:39:12.063569Z","shell.execute_reply.started":"2021-07-05T13:38:49.622157Z","shell.execute_reply":"2021-07-05T13:39:12.062754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_X = np.array(test_X)","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:39:12.064983Z","iopub.execute_input":"2021-07-05T13:39:12.065315Z","iopub.status.idle":"2021-07-05T13:39:12.156026Z","shell.execute_reply.started":"2021-07-05T13:39:12.065281Z","shell.execute_reply":"2021-07-05T13:39:12.155162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_X_rgb = np.repeat(test_X[..., np.newaxis], 3, -1)\nprint(test_X_rgb.shape)","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:39:12.157517Z","iopub.execute_input":"2021-07-05T13:39:12.157882Z","iopub.status.idle":"2021-07-05T13:39:12.898787Z","shell.execute_reply.started":"2021-07-05T13:39:12.157843Z","shell.execute_reply":"2021-07-05T13:39:12.897650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras import Model\nfrom tensorflow.keras.applications.vgg16 import VGG16\n\nmodel = tf.keras.applications.resnet50.ResNet50(include_top=False, weights='imagenet', input_shape=(128,128,3))\n#model = VGG16(include_top=False, weights='imagenet', input_shape=(256,256,3))\nx = Flatten() (model.output)\nx = Dense(32) (x)\nx = Dense(1, activation = 'sigmoid') (x)\n\nmodel = Model(inputs=model.inputs,outputs=x)","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:39:12.900233Z","iopub.execute_input":"2021-07-05T13:39:12.900627Z","iopub.status.idle":"2021-07-05T13:39:21.309161Z","shell.execute_reply.started":"2021-07-05T13:39:12.900588Z","shell.execute_reply":"2021-07-05T13:39:21.308330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:39:21.310446Z","iopub.execute_input":"2021-07-05T13:39:21.310791Z","iopub.status.idle":"2021-07-05T13:39:21.393294Z","shell.execute_reply.started":"2021-07-05T13:39:21.310756Z","shell.execute_reply":"2021-07-05T13:39:21.392423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# histories = []\n# losses = []\n# accuracies = []\n\nmodel.compile(optimizer=\"adam\", loss=\"binary_crossentropy\", metrics=[\"accuracy\"])\nmodel.fit(train_X_rgb, train_Y,  epochs=30, validation_split = 0.15)\nresults = model.evaluate(test_X_rgb, test_Y)\nresults = dict(zip(model.metrics_names,results))\n\n# histories.append(history)\n# accuracies.append(results['seg_seg_binary_accuracy'])    \n# losses.append(results['seg_loss'])","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:39:21.394480Z","iopub.execute_input":"2021-07-05T13:39:21.394815Z","iopub.status.idle":"2021-07-05T13:51:30.079121Z","shell.execute_reply.started":"2021-07-05T13:39:21.394783Z","shell.execute_reply":"2021-07-05T13:51:30.076417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(results)","metadata":{"execution":{"iopub.status.busy":"2021-07-05T13:51:30.084787Z","iopub.execute_input":"2021-07-05T13:51:30.085057Z","iopub.status.idle":"2021-07-05T13:51:30.093941Z","shell.execute_reply.started":"2021-07-05T13:51:30.085029Z","shell.execute_reply":"2021-07-05T13:51:30.093000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test_X_rgb[:10])","metadata":{"execution":{"iopub.status.busy":"2021-07-05T14:07:35.368403Z","iopub.execute_input":"2021-07-05T14:07:35.368739Z","iopub.status.idle":"2021-07-05T14:07:35.543904Z","shell.execute_reply.started":"2021-07-05T14:07:35.368688Z","shell.execute_reply":"2021-07-05T14:07:35.543071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred","metadata":{"execution":{"iopub.status.busy":"2021-07-05T14:07:42.869229Z","iopub.execute_input":"2021-07-05T14:07:42.869562Z","iopub.status.idle":"2021-07-05T14:07:42.876109Z","shell.execute_reply.started":"2021-07-05T14:07:42.869532Z","shell.execute_reply":"2021-07-05T14:07:42.875031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_Y[:10]","metadata":{"execution":{"iopub.status.busy":"2021-07-05T14:17:52.831200Z","iopub.execute_input":"2021-07-05T14:17:52.831531Z","iopub.status.idle":"2021-07-05T14:17:52.837509Z","shell.execute_reply.started":"2021-07-05T14:17:52.831494Z","shell.execute_reply":"2021-07-05T14:17:52.836608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(10):\n    if(test_Y[i]) == 1:\n        print(\"Label : PA\")\n    else:\n        print(\"Label : AP\")\n    \n    if(pred[i]) > 0.7:\n        print(\"Prediction : PA\")\n    else:\n        print(\"Prediction : AP\")  \n        \n    print(\"Test Image\")\n    plt.imshow(test_X_rgb[i])\n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-07-05T14:37:46.419115Z","iopub.execute_input":"2021-07-05T14:37:46.419436Z","iopub.status.idle":"2021-07-05T14:37:48.216579Z","shell.execute_reply.started":"2021-07-05T14:37:46.419406Z","shell.execute_reply":"2021-07-05T14:37:48.215574Z"},"trusted":true},"execution_count":null,"outputs":[]}]}