{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-22T15:05:49.001469Z","iopub.execute_input":"2021-06-22T15:05:49.001938Z","iopub.status.idle":"2021-06-22T15:05:49.007617Z","shell.execute_reply.started":"2021-06-22T15:05:49.001845Z","shell.execute_reply":"2021-06-22T15:05:49.006069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\n\nimport pydicom as dicom\nimport os\nimport cv2\nimport PIL # optional\nimport shutil","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:05:49.010361Z","iopub.execute_input":"2021-06-22T15:05:49.010879Z","iopub.status.idle":"2021-06-22T15:05:55.76626Z","shell.execute_reply.started":"2021-06-22T15:05:49.010834Z","shell.execute_reply":"2021-06-22T15:05:55.765303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading input data file in png format to dataframes\n#train images in png format and stored in annotation format in dataframe\nimport os\nkf = pd.DataFrame(columns=['data_type','StudyInstanceUID','image_file_ID','image_file_path'])\nfor dirname, _, filenames in os.walk('/kaggle/input/siim-covid19-detection-dicom-to-png-conversion'):\n    for filename in filenames:\n        \n        std_id = filename.split('_')[0]\n        #f_name_id = filename.split('_')[1].split('.')[0]\n        dat_type = dirname.split('/')[-1]\n        img_path = os.path.join(dirname, filename)\n        \n        kf.loc[kf.shape[0]] = [dat_type,std_id,filename,img_path]\n\n\npng_data = pd.DataFrame(kf.drop([0,1,2,3,4],axis=0).values,columns=kf.columns)\npng_data['image_file_ID'] = png_data.image_file_ID.str[13:-4]\npng_data\n\n# creating train and test datframes where all the image_study and studyInstanceUID are stored\npng_test = png_data[png_data['data_type'] == 'test']\npng_train = png_data[png_data['data_type'] == 'train']","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:05:55.768402Z","iopub.execute_input":"2021-06-22T15:05:55.768854Z","iopub.status.idle":"2021-06-22T15:06:22.797306Z","shell.execute_reply.started":"2021-06-22T15:05:55.76881Z","shell.execute_reply":"2021-06-22T15:06:22.796528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"png_train","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:22.800064Z","iopub.execute_input":"2021-06-22T15:06:22.800748Z","iopub.status.idle":"2021-06-22T15:06:22.824943Z","shell.execute_reply.started":"2021-06-22T15:06:22.800697Z","shell.execute_reply":"2021-06-22T15:06:22.824108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"png_test","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:22.826795Z","iopub.execute_input":"2021-06-22T15:06:22.827206Z","iopub.status.idle":"2021-06-22T15:06:22.842794Z","shell.execute_reply.started":"2021-06-22T15:06:22.827177Z","shell.execute_reply":"2021-06-22T15:06:22.841357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/siim-covid19-detection/sample_submission.csv\")\ntrain_image = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_image_level.csv\")\ntrain_study = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_study_level.csv\")\n\n# class label colum and case study id_number columns are created\n\ntrain_study['classification_class'] = np.zeros((train_study.shape[0],1))\nfor i in range(0,train_study.shape[0]):\n    train_study['classification_class'][i] = train_study.iloc[i].values[1:-1]\n\n# StudyInstanceUID \ntrain_study['StudyInstanceUID'] = train_study.id.str[0:-6]\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:22.844615Z","iopub.execute_input":"2021-06-22T15:06:22.845145Z","iopub.status.idle":"2021-06-22T15:06:27.12357Z","shell.execute_reply.started":"2021-06-22T15:06:22.8451Z","shell.execute_reply":"2021-06-22T15:06:27.122485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"itr = 0\ncl = []\nfor i in train_study['classification_class'].values:\n    if i[0]==1:\n        cl.append('Negative')\n    elif i[1]==1:\n        cl.append('Typical')\n    elif i[2]==1:\n        cl.append('Indeterminate')\n    elif i[3]==1:\n        cl.append('Atypical')\n    else:\n        a=1\n    itr = itr+1\n\ntrain_study['class_type'] = cl\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:27.127285Z","iopub.execute_input":"2021-06-22T15:06:27.127644Z","iopub.status.idle":"2021-06-22T15:06:27.166504Z","shell.execute_reply.started":"2021-06-22T15:06:27.127608Z","shell.execute_reply":"2021-06-22T15:06:27.16542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Selecting appropriate image for the StudyInstanceUID from coressponding multiple images_id's\n\nSIUID = train_image.groupby('StudyInstanceUID') #SIUID :- Syudy instance UID indexed dataframe\nim_fl_id = png_train.set_index('image_file_ID')\n\nImage_file_path = []\nImage_file_ID = []\nfor i in train_study['StudyInstanceUID'].values:\n    \n    pk = SIUID.get_group(i)\n    if (pk.shape[0] >1) & (pk.dropna(subset=['boxes']).shape[0]>0):\n        \n        im_id = SIUID.get_group(i).dropna(subset=['boxes']).values[0][0].split('_')[0]\n        path = im_fl_id.loc[im_id]['image_file_path']\n        Image_file_path.append(path)\n        Image_file_ID.append(im_id)\n    \n    else:\n        im_id = SIUID.get_group(i).values[0][0].split('_')[0]\n        path = im_fl_id.loc[im_id]['image_file_path']\n        Image_file_path.append(path)\n        Image_file_ID.append(im_id)\n\ntrain_study['image_file_path'] = Image_file_path\ntrain_study['image_file_ID'] = Image_file_ID\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:27.168392Z","iopub.execute_input":"2021-06-22T15:06:27.168886Z","iopub.status.idle":"2021-06-22T15:06:39.125114Z","shell.execute_reply.started":"2021-06-22T15:06:27.168837Z","shell.execute_reply":"2021-06-22T15:06:39.124318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using this information for classification problem alone\ntrain_study.info()","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:39.126497Z","iopub.execute_input":"2021-06-22T15:06:39.126885Z","iopub.status.idle":"2021-06-22T15:06:39.147704Z","shell.execute_reply.started":"2021-06-22T15:06:39.126845Z","shell.execute_reply":"2021-06-22T15:06:39.146914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This data frame is for the object localisation and classification combined activities\n\ntrain_Study_image = pd.merge(train_image,train_study,on='StudyInstanceUID',how=\"inner\")\n#\ntrain_Study_image.columns = ['Image_id', 'boxes', 'label', 'StudyInstanceUID', 'Study_id',\n       'Negative for Pneumonia', 'Typical Appearance',\n       'Indeterminate Appearance', 'Atypical Appearance',\n       'classification_class','class_type','image_file_path','image_file_ID']\n\ntrain_Study_image","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:39.148885Z","iopub.execute_input":"2021-06-22T15:06:39.149167Z","iopub.status.idle":"2021-06-22T15:06:39.189982Z","shell.execute_reply.started":"2021-06-22T15:06:39.149139Z","shell.execute_reply":"2021-06-22T15:06:39.189042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Study Instance ID where all the images of the study ID are labelled as none\ntrain_Study_image.set_index('StudyInstanceUID').loc['0d962bb4c9f2']","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:39.191229Z","iopub.execute_input":"2021-06-22T15:06:39.191489Z","iopub.status.idle":"2021-06-22T15:06:39.213816Z","shell.execute_reply.started":"2021-06-22T15:06:39.191461Z","shell.execute_reply":"2021-06-22T15:06:39.212824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pre-Trained Model Activation ","metadata":{}},{"cell_type":"code","source":"img_size = 256\nimg_depth = 3\nmodel = tf.keras.applications.resnet50.ResNet50(include_top=False, #Do not include FC layer at the end\n                                          input_shape=(img_size,img_size, img_depth),\n                                          weights='imagenet')","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:39.21505Z","iopub.execute_input":"2021-06-22T15:06:39.215329Z","iopub.status.idle":"2021-06-22T15:06:41.643812Z","shell.execute_reply.started":"2021-06-22T15:06:39.215293Z","shell.execute_reply":"2021-06-22T15:06:41.64263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Set pre-trained model layers to not trainable\nfor layer in model.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:41.645308Z","iopub.execute_input":"2021-06-22T15:06:41.645624Z","iopub.status.idle":"2021-06-22T15:06:41.655334Z","shell.execute_reply.started":"2021-06-22T15:06:41.645592Z","shell.execute_reply":"2021-06-22T15:06:41.654584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get Output layer of Pre0trained model\nx = model.output\n\n#Flatten the output to feed to Dense layer\nx = tf.keras.layers.Flatten()(x)\n\n#Add one Dense layer\nx = tf.keras.layers.Dense(200, activation='relu')(x)\n\n#Add output layer\nprediction = tf.keras.layers.Dense(4,activation='softmax')(x)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:41.657001Z","iopub.execute_input":"2021-06-22T15:06:41.657307Z","iopub.status.idle":"2021-06-22T15:06:41.886814Z","shell.execute_reply.started":"2021-06-22T15:06:41.65728Z","shell.execute_reply":"2021-06-22T15:06:41.885793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Using Keras Model class\nfinal_model = tf.keras.models.Model(inputs=model.input, #Pre-trained model input as input layer\n                                    outputs=prediction) #Output layer added","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:41.888181Z","iopub.execute_input":"2021-06-22T15:06:41.888469Z","iopub.status.idle":"2021-06-22T15:06:41.906717Z","shell.execute_reply.started":"2021-06-22T15:06:41.888441Z","shell.execute_reply":"2021-06-22T15:06:41.905792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:41.907831Z","iopub.execute_input":"2021-06-22T15:06:41.908167Z","iopub.status.idle":"2021-06-22T15:06:41.926274Z","shell.execute_reply.started":"2021-06-22T15:06:41.908138Z","shell.execute_reply":"2021-06-22T15:06:41.925221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image data augumentataion \ntrn_gen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255.,rotation_range=20,width_shift_range=0.2,height_shift_range=0.2,\n                                                shear_range=0.2,zoom_range=0.2,\n                                                fill_mode=\"nearest\",horizontal_flip=True,\n                                                #validation_split=0.2,\n                                                preprocessing_function=tf.keras.applications.resnet.preprocess_input)\n\n\nval_gen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255.,rotation_range=20,width_shift_range=0.2,height_shift_range=0.2,\n                                                shear_range=0.2,zoom_range=0.2,\n                                                fill_mode=\"nearest\",horizontal_flip=True,\n                                                #validation_split=0.2,\n                                                preprocessing_function=tf.keras.applications.resnet.preprocess_input)\n\ntest_gen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255.)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:41.927459Z","iopub.execute_input":"2021-06-22T15:06:41.927777Z","iopub.status.idle":"2021-06-22T15:06:41.935512Z","shell.execute_reply.started":"2021-06-22T15:06:41.927749Z","shell.execute_reply":"2021-06-22T15:06:41.934736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Just taking only the required columns from the trainstudy dataframe\nkj = train_study[['class_type','image_file_path']]\nkt = pd.get_dummies(train_study[['class_type']])\nfea = pd.concat([kt,kj],axis=1)\nfea.columns = ['Atypical', 'Indeterminate','Negative', 'Typical', 'class_type','image_file_path']\nfea","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:41.936977Z","iopub.execute_input":"2021-06-22T15:06:41.937284Z","iopub.status.idle":"2021-06-22T15:06:41.969428Z","shell.execute_reply.started":"2021-06-22T15:06:41.937229Z","shell.execute_reply":"2021-06-22T15:06:41.968638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_trn,X_val,Y_trn,Y_val = train_test_split(fea,fea[['class_type']],test_size=0.15,random_state=45)\n\n#Y_trn = tf.keras.utils.to_categorical(Y_trn, num_classes=4)\n#Y_val = tf.keras.utils.to_categorical(Y_val, num_classes=4)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:41.972689Z","iopub.execute_input":"2021-06-22T15:06:41.972992Z","iopub.status.idle":"2021-06-22T15:06:42.208309Z","shell.execute_reply.started":"2021-06-22T15:06:41.972963Z","shell.execute_reply":"2021-06-22T15:06:42.207597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_trn\nX_val","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:42.209573Z","iopub.execute_input":"2021-06-22T15:06:42.210031Z","iopub.status.idle":"2021-06-22T15:06:42.224221Z","shell.execute_reply.started":"2021-06-22T15:06:42.209985Z","shell.execute_reply":"2021-06-22T15:06:42.223308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create train and test generator\nbatchsize = 128\ntrain_generator =  trn_gen.flow_from_dataframe (X_trn,\n                                                x_col=\"image_file_path\",\n                                                y_col=['Atypical', 'Indeterminate','Negative', 'Typical'],\n                                                target_size=(256, 256),\n                                                color_mode=\"rgb\",\n                                            #classes=['Negative','Typical','Indeterminate','Atypical'], \n                                                class_mode=\"raw\",\n                                                batch_size=batchsize,\n                                                shuffle=True,\n                                                seed=20,\n                                                ) #batchsize can be changed\n\nval_generator = val_gen.flow_from_dataframe (X_val,\n                                            x_col=\"image_file_path\",\n                                            y_col=['Atypical', 'Indeterminate','Negative', 'Typical'],\n                                            target_size=(256, 256),\n                                            color_mode=\"rgb\",\n                                            #classes=['Negative','Typical','Indeterminate','Atypical'], \n                                            class_mode=\"raw\",\n                                            batch_size=batchsize,\n                                            shuffle=True,\n                                            seed=20,\n                                            )","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:42.225636Z","iopub.execute_input":"2021-06-22T15:06:42.225902Z","iopub.status.idle":"2021-06-22T15:06:44.358979Z","shell.execute_reply.started":"2021-06-22T15:06:42.225876Z","shell.execute_reply":"2021-06-22T15:06:44.35819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batchsize = 256\nfinal_model.fit_generator(train_generator, \n                          epochs=10,\n                          steps_per_epoch= X_trn.shape[0]//batchsize,\n                          validation_data=val_generator,\n                          validation_steps = X_val.shape[0]//batchsize,\n                          verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T15:06:44.360485Z","iopub.execute_input":"2021-06-22T15:06:44.360795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batchsize = 256\nfinal_model.fit_generator(train_generator, \n                          epochs=10,\n                          steps_per_epoch= X_trn.shape[0]//batchsize,\n                          validation_data=val_generator,\n                          validation_steps = X_val.shape[0]//batchsize,\n                          verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batchsize = 200\nfinal_model.fit_generator(train_generator, \n                          epochs=10,\n                          steps_per_epoch= X_trn.shape[0]//batchsize,\n                          validation_data=val_generator,\n                          validation_steps = X_val.shape[0]//batchsize,\n                          verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batchsize = 100\nfinal_model.fit_generator(train_generator, \n                          epochs=10,\n                          steps_per_epoch= X_trn.shape[0]//batchsize,\n                          validation_data=val_generator,\n                          validation_steps = X_val.shape[0]//batchsize,\n                          verbose=10)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nprint(len(model.layers))\nfor layer in model.layers[150:]:\n    layer.trainable =  True \n'''","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}