{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-25T17:08:35.658391Z","iopub.execute_input":"2021-06-25T17:08:35.658815Z","iopub.status.idle":"2021-06-25T17:08:35.670496Z","shell.execute_reply.started":"2021-06-25T17:08:35.658715Z","shell.execute_reply":"2021-06-25T17:08:35.669707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\n\nimport pydicom as dicom\nimport os\nimport cv2\nimport PIL # optional\nimport tensorflow as tf\nimport shutil","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:08:46.789327Z","iopub.execute_input":"2021-06-25T17:08:46.789698Z","iopub.status.idle":"2021-06-25T17:08:53.098353Z","shell.execute_reply.started":"2021-06-25T17:08:46.789666Z","shell.execute_reply":"2021-06-25T17:08:53.097219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nkf = pd.DataFrame(columns=['data_set_type','StudyInstanceUID','folder_level_2','image_file_name','image_file_path'])\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        kf.loc[kf.shape[0]] = [dirname.split('/')[-3], dirname.split('/')[-2],dirname.split('/')[-1],filename,os.path.join(dirname, filename)]\n\ndf = pd.DataFrame(kf.drop([0,1,2],axis=0).values,columns=kf.columns)\ndf","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:08:53.099602Z","iopub.execute_input":"2021-06-25T17:08:53.099865Z","iopub.status.idle":"2021-06-25T17:09:56.165688Z","shell.execute_reply.started":"2021-06-25T17:08:53.09984Z","shell.execute_reply":"2021-06-25T17:09:56.164872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['StudyInstanceUID'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:15:38.488986Z","iopub.execute_input":"2021-06-25T17:15:38.489317Z","iopub.status.idle":"2021-06-25T17:15:38.506063Z","shell.execute_reply.started":"2021-06-25T17:15:38.489275Z","shell.execute_reply":"2021-06-25T17:15:38.505098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking if all the images were captured in the dataFrame\ndf.set_index('data_set_type').loc['train'].set_index('StudyInstanceUID').loc['a4e94133d95a']\n\n# A random studyInstanceUID is cheked if the all images inside the subfolder are captured","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:15:00.550249Z","iopub.execute_input":"2021-06-25T17:15:00.550636Z","iopub.status.idle":"2021-06-25T17:15:00.572989Z","shell.execute_reply.started":"2021-06-25T17:15:00.550603Z","shell.execute_reply":"2021-06-25T17:15:00.571969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Dataset Snippet:\n    The train dataset comprises 6,334 chest scans in DICOM format","metadata":{}},{"cell_type":"markdown","source":"# Look at the train data given","metadata":{}},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/siim-covid19-detection/sample_submission.csv\")\ntrain_image = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_image_level.csv\")\ntrain_study = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_study_level.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:16:21.649345Z","iopub.execute_input":"2021-06-25T17:16:21.649696Z","iopub.status.idle":"2021-06-25T17:16:21.725366Z","shell.execute_reply.started":"2021-06-25T17:16:21.649666Z","shell.execute_reply":"2021-06-25T17:16:21.724394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:16:27.828138Z","iopub.execute_input":"2021-06-25T17:16:27.828508Z","iopub.status.idle":"2021-06-25T17:16:27.842963Z","shell.execute_reply.started":"2021-06-25T17:16:27.828475Z","shell.execute_reply":"2021-06-25T17:16:27.841789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unique values present in the train images with with respect to case study ID\ntrain_image['StudyInstanceUID'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:17:16.744787Z","iopub.execute_input":"2021-06-25T17:17:16.74515Z","iopub.status.idle":"2021-06-25T17:17:16.757319Z","shell.execute_reply.started":"2021-06-25T17:17:16.745121Z","shell.execute_reply":"2021-06-25T17:17:16.756396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Certain case study Id contain more than one images.","metadata":{}},{"cell_type":"code","source":"# this is case study id with 5 image study files\ntrain_image.set_index('StudyInstanceUID').loc['a4e94133d95a']","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:17:25.104322Z","iopub.execute_input":"2021-06-25T17:17:25.104824Z","iopub.status.idle":"2021-06-25T17:17:25.119691Z","shell.execute_reply.started":"2021-06-25T17:17:25.104796Z","shell.execute_reply":"2021-06-25T17:17:25.118806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# this is case study ID with 9 image files\ntrain_image.set_index('StudyInstanceUID').loc['0fd2db233deb']","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:17:29.961049Z","iopub.execute_input":"2021-06-25T17:17:29.961422Z","iopub.status.idle":"2021-06-25T17:17:29.975743Z","shell.execute_reply.started":"2021-06-25T17:17:29.961383Z","shell.execute_reply":"2021-06-25T17:17:29.974692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_study","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:18:20.027209Z","iopub.execute_input":"2021-06-25T17:18:20.027591Z","iopub.status.idle":"2021-06-25T17:18:20.042484Z","shell.execute_reply.started":"2021-06-25T17:18:20.027559Z","shell.execute_reply":"2021-06-25T17:18:20.041544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class label colum and case study id_number columns are created\n\ntrain_study['classification_class'] = np.zeros((train_study.shape[0],1))\nfor i in range(0,train_study.shape[0]):\n    train_study['classification_class'][i] = train_study.iloc[i].values[1:-1]\n\n\ntrain_study['id_num_only'] = train_study.id.str[:-6]    \ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:19:01.275138Z","iopub.execute_input":"2021-06-25T17:19:01.275513Z","iopub.status.idle":"2021-06-25T17:19:05.084612Z","shell.execute_reply.started":"2021-06-25T17:19:01.275484Z","shell.execute_reply":"2021-06-25T17:19:05.083866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking the classification label \n# this label will be used for the traning purpose \ntrain_study['classification_class'][50]","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:22:32.236827Z","iopub.execute_input":"2021-06-25T17:22:32.237234Z","iopub.status.idle":"2021-06-25T17:22:32.243612Z","shell.execute_reply.started":"2021-06-25T17:22:32.237201Z","shell.execute_reply":"2021-06-25T17:22:32.242711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Varifying if the Case study id and studyinstanceUID are having same values\n\ns1 = set(train_study['id_num_only'].value_counts().keys())\ns2 = set(train_image['StudyInstanceUID'].values)\n\n# difference between them is null set => every thing is in common\nprint(\"set intersection: - \",s1-s2)\n# even the length is same\nprint(\"set len difference: - \", len(s1) - len(s2))","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:22:40.944812Z","iopub.execute_input":"2021-06-25T17:22:40.945183Z","iopub.status.idle":"2021-06-25T17:22:40.961117Z","shell.execute_reply.started":"2021-06-25T17:22:40.945148Z","shell.execute_reply":"2021-06-25T17:22:40.960014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Intersection is null set and len is of same size, the difference in two lengths is 0\n\nThis shows that the Case Study ID given in train study and StudyInstanceUID are the same","metadata":{}},{"cell_type":"code","source":"train_image.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:24:00.569803Z","iopub.execute_input":"2021-06-25T17:24:00.570557Z","iopub.status.idle":"2021-06-25T17:24:00.592085Z","shell.execute_reply.started":"2021-06-25T17:24:00.570485Z","shell.execute_reply":"2021-06-25T17:24:00.590865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Look out for boxes. The number of boxes are very much less compared to number of images and they are also lessthan number of case study ID's. Those values are considered as null values or no observation from those images were made.","metadata":{}},{"cell_type":"code","source":"# using this information for classification problem alone\ntrain_study.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-25T17:24:28.90519Z","iopub.execute_input":"2021-06-25T17:24:28.90558Z","iopub.status.idle":"2021-06-25T17:24:28.925106Z","shell.execute_reply.started":"2021-06-25T17:24:28.905546Z","shell.execute_reply":"2021-06-25T17:24:28.923889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Converting DICOM format to image format (jpg / png format)\n\nWe can't use the dicom image format directly for image processing in deep Learning  ","metadata":{}},{"cell_type":"markdown","source":"Clearly shows that 'id_num' and 'StudyInstanceUID' same and have unique values matching","metadata":{}},{"cell_type":"markdown","source":"# Batch wise conversion of dicom images to png format","metadata":{}},{"cell_type":"code","source":"import os\nimport SimpleITK\nimport cv2\nfrom tqdm import tqdm\nimport shutil\n \ndef convert_from_dicom_to_jpg(dcm_image_path,output_jpg_path):     #,output_jpg_path):\n           \n    ds_array = SimpleITK.ReadImage(dcm_image_path)  \n    img_array = SimpleITK.GetArrayFromImage(ds_array)  \n            \n    shape = img_array.shape\n    img_array = np.reshape(img_array, (shape[1], shape[2]))  \n    high = np.max(img_array)\n    low = np.min(img_array)\n        \n    lungwin = np.array([low*1.,high*1.])\n    newimg = (img_array-lungwin[0])/(lungwin[1]-lungwin[0])\n    newimg = (newimg*255).astype('uint8')\n    stacked_img = np.stack((newimg,) * 3, axis=-1)\n    x_imag = tf.keras.preprocessing.image.array_to_img(stacked_img)\n    \n    # required image format and size as a input to Deep learning network\n    \n    ht = tf.keras.preprocessing.image.img_to_array(x_imag.resize([256,256]))\n    cv2.imwrite(output_jpg_path, ht, [int(cv2.IMWRITE_JPEG_QUALITY), 10])\n    #return ht","metadata":{"execution":{"iopub.status.busy":"2021-06-13T04:40:00.864239Z","iopub.execute_input":"2021-06-13T04:40:00.864577Z","iopub.status.idle":"2021-06-13T04:40:01.403776Z","shell.execute_reply.started":"2021-06-13T04:40:00.864552Z","shell.execute_reply":"2021-06-13T04:40:01.403072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"itr = 0\n\nif os.path.exists('/kaggle/working/train/') == True:\n    #shutil.rmtree('/kaggle/working/train/')\nif os.path.exists('/kaggle/working/test/') == True:\n    #shutil.rmtree('/kaggle/working/test/')\n        \nos.makedirs('/kaggle/working/test/')\nos.makedirs('/kaggle/working/train/')","metadata":{"execution":{"iopub.status.busy":"2021-06-12T10:19:31.939463Z","iopub.execute_input":"2021-06-12T10:19:31.939819Z","iopub.status.idle":"2021-06-12T10:30:34.306675Z","shell.execute_reply.started":"2021-06-12T10:19:31.939789Z","shell.execute_reply":"2021-06-12T10:30:34.305072Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Flow from Data frame to easy the batch generation process","metadata":{}},{"cell_type":"code","source":"# code for flow from dataframe using keras API\n\ntrain_generator=datagen.flow_from_dataframe(\ndataframe=traindf,\ndirectory=\"./train/\",\nx_col=\"id\",\ny_col=\"label\",\nsubset=\"training\",\nbatch_size=32,\nseed=42,\nshuffle=True,\nclass_mode=\"categorical\",\ntarget_size=(32,32))\nvalid_generator=datagen.flow_from_dataframe(\ndataframe=traindf,\ndirectory=\"./train/\",\nx_col=\"id\",\ny_col=\"label\",\nsubset=\"validation\",\nbatch_size=32,\nseed=42,\nshuffle=True,\nclass_mode=\"categorical\",\ntarget_size=(32,32))\ntest_datagen=ImageDataGenerator(rescale=1./255.)\ntest_generator=test_datagen.flow_from_dataframe(\ndataframe=testdf,\ndirectory=\"./test/\",\nx_col=\"id\",\ny_col=None,\nbatch_size=32,\nseed=42,\nshuffle=False,\nclass_mode=None,\ntarget_size=(32,32))","metadata":{"execution":{"iopub.status.busy":"2021-06-20T08:29:08.153373Z","iopub.execute_input":"2021-06-20T08:29:08.153975Z","iopub.status.idle":"2021-06-20T08:29:08.209693Z","shell.execute_reply.started":"2021-06-20T08:29:08.153882Z","shell.execute_reply":"2021-06-20T08:29:08.20753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batbatch_generator.counter = 0\n\ndef batch_generator(df, batchsize=32,train_mode = True):\n    \n  \n    # Data agumentation batch wise through for model.fit_generator function\n    \n    img_generator= tf.keras.preprocessing.image.ImageDataGenerator(rotation_range=20,\n                                                                   width_shift_range=0.2,\n                                                                   height_shift_range=0.2,\n                                                                   horizontal_flip=True)   \n\n    \n    while True:       \n                \n        #Generate random numbers to pick images from dataset\n        batch_nums = np.random.randint(0,df.shape[0],batchsize)\n        \n        #Initialize batch images array\n        batch_images = np.zeros((batchsize,img_size, img_size,img_depth))\n        \n        #Initiate batch label array\n        batch_labels = np.zeros((batchsize, len(class_names)))\n        \n        df_batch = []\n        for i in batch_nums:\n            \n            dcm_image_path = df['image_file_path'].values[i]   \n            rt = convert_from_dicom_to_jpg(dcm_image_path)\n            df_batch.append(rt)\n               \n        for i in np.arange(batchsize):\n\n            #Load image in the array format\n            flower_image = df_batch[i]\n            #Convert class to one hot encoding\n            img_class = \n            \n            if train_mode = True:\n                #Apply transform\n                flower_image =  img_generator.random_transform(flower_image)\n            else:\n                a1 = 1 # random call nothing to do with main code we can use #pass\n\n            #Update batch images and class arrays\n            batch_images[i] = flower_image\n            batch_labels[i] = img_class\n\n        \n        batch_images = tf.keras.applications.resnet50.preprocess_input(batch_images)\n        \n        yield batch_images, batch_labels         ","metadata":{"execution":{"iopub.status.busy":"2021-06-11T08:14:09.085568Z","iopub.status.idle":"2021-06-11T08:14:09.086374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.set_index('data_set_type').loc['test']","metadata":{"execution":{"iopub.status.busy":"2021-06-13T05:02:02.671607Z","iopub.execute_input":"2021-06-13T05:02:02.671925Z","iopub.status.idle":"2021-06-13T05:02:02.691131Z","shell.execute_reply.started":"2021-06-13T05:02:02.6719Z","shell.execute_reply":"2021-06-13T05:02:02.690162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pre-Trained Model Activation ","metadata":{}},{"cell_type":"code","source":"model = tf.keras.applications.resnet50.ResNet50(include_top=False, #Do not include FC layer at the end\n                                          input_shape=(img_size,img_size, img_depth),\n                                          weights='imagenet')","metadata":{"execution":{"iopub.status.busy":"2021-06-11T09:43:28.878523Z","iopub.execute_input":"2021-06-11T09:43:28.879002Z","iopub.status.idle":"2021-06-11T09:43:28.884514Z","shell.execute_reply.started":"2021-06-11T09:43:28.87896Z","shell.execute_reply":"2021-06-11T09:43:28.88331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Set pre-trained model layers to not trainable\nfor layer in model.layers:\n    layer.trainable = False","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get Output layer of Pre0trained model\nx = model.output\n\n#Flatten the output to feed to Dense layer\nx = tf.keras.layers.Flatten()(x)\n\n#Add one Dense layer\nx = tf.keras.layers.Dense(200, activation='relu')(x)\n\n#Add output layer\nprediction = tf.keras.layers.Dense(len(class_names),activation='softmax')(x)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Using Keras Model class\nfinal_model = tf.keras.models.Model(inputs=model.input, #Pre-trained model input as input layer\n                                    outputs=prediction) #Output layer added","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create train and test generator\nbatchsize = 64\ntrain_generator = batch_generator(train_df, batchsize=batchsize) #batchsize can be changed\ntest_generator = batch_generator(test_df, batchsize=batchsize, train_mode=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model.fit_generator(train_generator, \n                          epochs=10,\n                          steps_per_epoch= train_df.shape[0]//batchsize,\n                          validation_data=test_generator,\n                          validation_steps = test_df.shape[0]//batchsize)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(model.layers))\nfor layer in model.layers[150:]:\n    layer.trainable =  True ","metadata":{},"execution_count":null,"outputs":[]}]}