{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-23T04:14:08.616213Z","iopub.execute_input":"2021-06-23T04:14:08.616776Z","iopub.status.idle":"2021-06-23T04:14:08.630403Z","shell.execute_reply.started":"2021-06-23T04:14:08.61659Z","shell.execute_reply":"2021-06-23T04:14:08.62927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\n\nimport pydicom as dicom\nimport os\nimport cv2\nimport PIL # optional\nimport shutil","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:14:08.632383Z","iopub.execute_input":"2021-06-23T04:14:08.632959Z","iopub.status.idle":"2021-06-23T04:14:15.541599Z","shell.execute_reply.started":"2021-06-23T04:14:08.632903Z","shell.execute_reply":"2021-06-23T04:14:15.540548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading input data file in png format to dataframes\n#train images in png format and stored in annotation format in dataframe\nimport os\nkf = pd.DataFrame(columns=['data_type','StudyInstanceUID','image_file_ID','image_file_path'])\nfor dirname, _, filenames in os.walk('/kaggle/input/siim-covid19-detection-dicom-to-png-conversion'):\n    for filename in filenames:\n        \n        std_id = filename.split('_')[0]\n        #f_name_id = filename.split('_')[1].split('.')[0]\n        dat_type = dirname.split('/')[-1]\n        img_path = os.path.join(dirname, filename)\n        \n        kf.loc[kf.shape[0]] = [dat_type,std_id,filename,img_path]\n\n\npng_data = pd.DataFrame(kf.drop([0,1,2,3,4],axis=0).values,columns=kf.columns)\npng_data['image_file_ID'] = png_data.image_file_ID.str[13:-4]\npng_data\n\n# creating train and test datframes where all the image_study and studyInstanceUID are stored\npng_test = png_data[png_data['data_type'] == 'test']\npng_train = png_data[png_data['data_type'] == 'train']","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:14:15.544042Z","iopub.execute_input":"2021-06-23T04:14:15.544377Z","iopub.status.idle":"2021-06-23T04:14:47.868514Z","shell.execute_reply.started":"2021-06-23T04:14:15.544348Z","shell.execute_reply":"2021-06-23T04:14:47.86746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"png_train","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:14:47.870549Z","iopub.execute_input":"2021-06-23T04:14:47.871013Z","iopub.status.idle":"2021-06-23T04:14:47.895418Z","shell.execute_reply.started":"2021-06-23T04:14:47.870969Z","shell.execute_reply":"2021-06-23T04:14:47.893629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"png_test","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:14:47.897522Z","iopub.execute_input":"2021-06-23T04:14:47.897961Z","iopub.status.idle":"2021-06-23T04:14:47.916523Z","shell.execute_reply.started":"2021-06-23T04:14:47.897921Z","shell.execute_reply":"2021-06-23T04:14:47.915373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/siim-covid19-detection/sample_submission.csv\")\ntrain_image = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_image_level.csv\")\ntrain_study = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_study_level.csv\")\n\n# class label colum and case study id_number columns are created\n\ntrain_study['classification_class'] = np.zeros((train_study.shape[0],1))\nfor i in range(0,train_study.shape[0]):\n    train_study['classification_class'][i] = train_study.iloc[i].values[1:-1]\n\n# StudyInstanceUID \ntrain_study['StudyInstanceUID'] = train_study.id.str[0:-6]\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:14:47.917993Z","iopub.execute_input":"2021-06-23T04:14:47.918602Z","iopub.status.idle":"2021-06-23T04:14:49.921784Z","shell.execute_reply.started":"2021-06-23T04:14:47.918561Z","shell.execute_reply":"2021-06-23T04:14:49.92073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"itr = 0\ncl = []\nfor i in train_study['classification_class'].values:\n    if i[0]==1:\n        cl.append('Negative')\n    elif i[1]==1:\n        cl.append('Typical')\n    elif i[2]==1:\n        cl.append('Indeterminate')\n    elif i[3]==1:\n        cl.append('Atypical')\n    else:\n        a=1\n    itr = itr+1\n\ntrain_study['class_type'] = cl\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:14:49.92338Z","iopub.execute_input":"2021-06-23T04:14:49.92394Z","iopub.status.idle":"2021-06-23T04:14:49.965256Z","shell.execute_reply.started":"2021-06-23T04:14:49.923899Z","shell.execute_reply":"2021-06-23T04:14:49.964007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Selecting appropriate image for the StudyInstanceUID from coressponding multiple images_id's\n\nSIUID = train_image.groupby('StudyInstanceUID') #SIUID :- Syudy instance UID indexed dataframe\nim_fl_id = png_train.set_index('image_file_ID')\n\nImage_file_path = []\nImage_file_ID = []\nfor i in train_study['StudyInstanceUID'].values:\n    \n    pk = SIUID.get_group(i)\n    if (pk.shape[0] >1) & (pk.dropna(subset=['boxes']).shape[0]>0):\n        \n        im_id = SIUID.get_group(i).dropna(subset=['boxes']).values[0][0].split('_')[0]\n        path = im_fl_id.loc[im_id]['image_file_path']\n        Image_file_path.append(path)\n        Image_file_ID.append(im_id)\n    \n    else:\n        im_id = SIUID.get_group(i).values[0][0].split('_')[0]\n        path = im_fl_id.loc[im_id]['image_file_path']\n        Image_file_path.append(path)\n        Image_file_ID.append(im_id)\n\ntrain_study['image_file_path'] = Image_file_path\ntrain_study['image_file_ID'] = Image_file_ID\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:14:49.966838Z","iopub.execute_input":"2021-06-23T04:14:49.967245Z","iopub.status.idle":"2021-06-23T04:15:04.318934Z","shell.execute_reply.started":"2021-06-23T04:14:49.967206Z","shell.execute_reply":"2021-06-23T04:15:04.318008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using this information for classification problem alone\ntrain_study.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:04.322676Z","iopub.execute_input":"2021-06-23T04:15:04.323006Z","iopub.status.idle":"2021-06-23T04:15:04.344945Z","shell.execute_reply.started":"2021-06-23T04:15:04.322966Z","shell.execute_reply":"2021-06-23T04:15:04.343584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This data frame is for the object localisation and classification combined activities\n\ntrain_Study_image = pd.merge(train_image,train_study,on='StudyInstanceUID',how=\"inner\")\n#\ntrain_Study_image.columns = ['Image_id', 'boxes', 'label', 'StudyInstanceUID', 'Study_id',\n       'Negative for Pneumonia', 'Typical Appearance',\n       'Indeterminate Appearance', 'Atypical Appearance',\n       'classification_class','class_type','image_file_path','image_file_ID']\n\ntrain_Study_image","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:04.347723Z","iopub.execute_input":"2021-06-23T04:15:04.348144Z","iopub.status.idle":"2021-06-23T04:15:04.392724Z","shell.execute_reply.started":"2021-06-23T04:15:04.348106Z","shell.execute_reply":"2021-06-23T04:15:04.391487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Study Instance ID where all the images of the study ID are labelled as none\ntrain_Study_image.set_index('StudyInstanceUID').loc['0d962bb4c9f2']","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:04.394288Z","iopub.execute_input":"2021-06-23T04:15:04.394739Z","iopub.status.idle":"2021-06-23T04:15:04.420662Z","shell.execute_reply.started":"2021-06-23T04:15:04.394685Z","shell.execute_reply":"2021-06-23T04:15:04.4194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pre-Trained Model Activation ","metadata":{}},{"cell_type":"code","source":"img_size = 256\nimg_depth = 3\nmodel = tf.keras.applications.EfficientNetB6(include_top=False, #Do not include FC layer at the end\n                                          input_shape=(img_size,img_size, img_depth),\n                                          weights='imagenet')","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:04.422233Z","iopub.execute_input":"2021-06-23T04:15:04.422688Z","iopub.status.idle":"2021-06-23T04:15:16.842585Z","shell.execute_reply.started":"2021-06-23T04:15:04.422635Z","shell.execute_reply":"2021-06-23T04:15:16.84148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Set pre-trained model layers to not trainable\nfor layer in model.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:16.845064Z","iopub.execute_input":"2021-06-23T04:15:16.845472Z","iopub.status.idle":"2021-06-23T04:15:16.875944Z","shell.execute_reply.started":"2021-06-23T04:15:16.845429Z","shell.execute_reply":"2021-06-23T04:15:16.874732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get Output layer of Pre0trained model\nx = model.output\n\n#Flatten the output to feed to Dense layer\nx = tf.keras.layers.Flatten()(x)\n\n#Add one Dense layer\nx = tf.keras.layers.Dense(200, activation='relu')(x)\n\n#Add output layer\nprediction = tf.keras.layers.Dense(4,activation='softmax')(x)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:16.877753Z","iopub.execute_input":"2021-06-23T04:15:16.878308Z","iopub.status.idle":"2021-06-23T04:15:16.910203Z","shell.execute_reply.started":"2021-06-23T04:15:16.878267Z","shell.execute_reply":"2021-06-23T04:15:16.909235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Using Keras Model class\nfinal_model = tf.keras.models.Model(inputs=model.input, #Pre-trained model input as input layer\n                                    outputs=prediction) #Output layer added","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:16.911449Z","iopub.execute_input":"2021-06-23T04:15:16.911838Z","iopub.status.idle":"2021-06-23T04:15:16.960393Z","shell.execute_reply.started":"2021-06-23T04:15:16.911784Z","shell.execute_reply":"2021-06-23T04:15:16.959452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:16.961593Z","iopub.execute_input":"2021-06-23T04:15:16.962159Z","iopub.status.idle":"2021-06-23T04:15:16.989667Z","shell.execute_reply.started":"2021-06-23T04:15:16.962119Z","shell.execute_reply":"2021-06-23T04:15:16.988802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image data augumentataion \ntrn_gen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255.,rotation_range=20,width_shift_range=0.2,height_shift_range=0.2,\n                                                shear_range=0.2,zoom_range=0.2,\n                                                fill_mode=\"nearest\",horizontal_flip=True,\n                                                #validation_split=0.2,\n                                                preprocessing_function=tf.keras.applications.efficientnet.preprocess_input)\n\n\nval_gen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255.,rotation_range=20,width_shift_range=0.2,height_shift_range=0.2,\n                                                shear_range=0.2,zoom_range=0.2,\n                                                fill_mode=\"nearest\",horizontal_flip=True,\n                                                #validation_split=0.2,\n                                                preprocessing_function=tf.keras.applications.efficientnet.preprocess_input)\n\ntest_gen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255.)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:16.99113Z","iopub.execute_input":"2021-06-23T04:15:16.991526Z","iopub.status.idle":"2021-06-23T04:15:17.000039Z","shell.execute_reply.started":"2021-06-23T04:15:16.991481Z","shell.execute_reply":"2021-06-23T04:15:16.998589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Just taking only the required columns from the trainstudy dataframe\nkj = train_study[['class_type','image_file_path']]\nkt = pd.get_dummies(train_study[['class_type']])\nfea = pd.concat([kt,kj],axis=1)\nfea.columns = ['Atypical', 'Indeterminate','Negative', 'Typical', 'class_type','image_file_path']\nfea","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:17.002113Z","iopub.execute_input":"2021-06-23T04:15:17.00287Z","iopub.status.idle":"2021-06-23T04:15:17.038784Z","shell.execute_reply.started":"2021-06-23T04:15:17.002813Z","shell.execute_reply":"2021-06-23T04:15:17.037845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_trn,X_val,Y_trn,Y_val = train_test_split(fea,fea[['class_type']],test_size=0.15,random_state=45)\n\n#Y_trn = tf.keras.utils.to_categorical(Y_trn, num_classes=4)\n#Y_val = tf.keras.utils.to_categorical(Y_val, num_classes=4)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:17.040275Z","iopub.execute_input":"2021-06-23T04:15:17.040709Z","iopub.status.idle":"2021-06-23T04:15:17.250552Z","shell.execute_reply.started":"2021-06-23T04:15:17.04067Z","shell.execute_reply":"2021-06-23T04:15:17.249478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_trn\nX_val","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:17.254002Z","iopub.execute_input":"2021-06-23T04:15:17.254303Z","iopub.status.idle":"2021-06-23T04:15:17.275961Z","shell.execute_reply.started":"2021-06-23T04:15:17.254276Z","shell.execute_reply":"2021-06-23T04:15:17.274302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create train and test generator\nbatchsize = 128\ntrain_generator =  trn_gen.flow_from_dataframe (X_trn,\n                                                x_col=\"image_file_path\",\n                                                y_col=['Atypical', 'Indeterminate','Negative', 'Typical'],\n                                                target_size=(256, 256),\n                                                color_mode=\"rgb\",\n                                            #classes=['Negative','Typical','Indeterminate','Atypical'], \n                                                class_mode=\"raw\",\n                                                batch_size=batchsize,\n                                                shuffle=True,\n                                                seed=20,\n                                                ) #batchsize can be changed\n\nval_generator = val_gen.flow_from_dataframe (X_val,\n                                            x_col=\"image_file_path\",\n                                            y_col=['Atypical', 'Indeterminate','Negative', 'Typical'],\n                                            target_size=(256, 256),\n                                            color_mode=\"rgb\",\n                                            #classes=['Negative','Typical','Indeterminate','Atypical'], \n                                            class_mode=\"raw\",\n                                            batch_size=batchsize,\n                                            shuffle=True,\n                                            seed=20,\n                                            )","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:17.277942Z","iopub.execute_input":"2021-06-23T04:15:17.278469Z","iopub.status.idle":"2021-06-23T04:15:18.79045Z","shell.execute_reply.started":"2021-06-23T04:15:17.278427Z","shell.execute_reply":"2021-06-23T04:15:18.78888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batchsize = 256\nfinal_model.fit_generator(train_generator, \n                          epochs=10,\n                          steps_per_epoch= X_trn.shape[0]//batchsize,\n                          validation_data=val_generator,\n                          validation_steps = X_val.shape[0]//batchsize,\n                          verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:15:18.793735Z","iopub.execute_input":"2021-06-23T04:15:18.794053Z","iopub.status.idle":"2021-06-23T04:26:22.943607Z","shell.execute_reply.started":"2021-06-23T04:15:18.794025Z","shell.execute_reply":"2021-06-23T04:26:22.94253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batchsize = 200\nfinal_model.fit_generator(train_generator, \n                          epochs=30,\n                          steps_per_epoch= X_trn.shape[0]//batchsize,\n                          validation_data=val_generator,\n                          validation_steps = X_val.shape[0]//batchsize,\n                          verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T04:26:22.947715Z","iopub.execute_input":"2021-06-23T04:26:22.948019Z","iopub.status.idle":"2021-06-23T05:04:35.792571Z","shell.execute_reply.started":"2021-06-23T04:26:22.947989Z","shell.execute_reply":"2021-06-23T05:04:35.791504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batchsize = 150\nfinal_model.fit_generator(train_generator, \n                          epochs=30,\n                          steps_per_epoch= X_trn.shape[0]//batchsize,\n                          validation_data=val_generator,\n                          validation_steps = X_val.shape[0]//batchsize,\n                          verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T05:04:35.794211Z","iopub.execute_input":"2021-06-23T05:04:35.794618Z","iopub.status.idle":"2021-06-23T05:56:36.585204Z","shell.execute_reply.started":"2021-06-23T05:04:35.794576Z","shell.execute_reply":"2021-06-23T05:56:36.584184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batchsize = 100\nfinal_model.fit_generator(train_generator, \n                          epochs=50,\n                          steps_per_epoch= X_trn.shape[0]//batchsize,\n                          validation_data=val_generator,\n                          validation_steps = X_val.shape[0]//batchsize,\n                          verbose=10)","metadata":{"execution":{"iopub.status.busy":"2021-06-23T05:56:36.588366Z","iopub.execute_input":"2021-06-23T05:56:36.588653Z","iopub.status.idle":"2021-06-23T05:58:41.840672Z","shell.execute_reply.started":"2021-06-23T05:56:36.588611Z","shell.execute_reply":"2021-06-23T05:58:41.839564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nprint(len(model.layers))\nfor layer in model.layers[150:]:\n    layer.trainable =  True \n'''","metadata":{"execution":{"iopub.status.busy":"2021-06-23T05:58:41.842612Z","iopub.execute_input":"2021-06-23T05:58:41.843309Z","iopub.status.idle":"2021-06-23T05:58:41.851116Z","shell.execute_reply.started":"2021-06-23T05:58:41.843255Z","shell.execute_reply":"2021-06-23T05:58:41.849619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}