{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"## This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n        \n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-24T04:30:32.829434Z","iopub.execute_input":"2021-07-24T04:30:32.830013Z","iopub.status.idle":"2021-07-24T04:30:32.839347Z","shell.execute_reply.started":"2021-07-24T04:30:32.829913Z","shell.execute_reply":"2021-07-24T04:30:32.838344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\n\nimport pydicom as dicom\nimport os\nimport cv2\nimport PIL # optional\nimport shutil","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:30:32.852754Z","iopub.execute_input":"2021-07-24T04:30:32.853291Z","iopub.status.idle":"2021-07-24T04:30:40.60809Z","shell.execute_reply.started":"2021-07-24T04:30:32.853256Z","shell.execute_reply":"2021-07-24T04:30:40.606963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading input data file in png format to dataframes\n#train images in png format and stored in annotation format in dataframe\nimport os\nkf = pd.DataFrame(columns=['data_type','StudyInstanceUID','image_file_ID','image_file_path'])\nfor dirname, _, filenames in os.walk('/kaggle/input/siim-covid19-detection-dicom-to-png-conversion'):\n    for filename in filenames:\n        \n        std_id = filename.split('_')[0]\n        #f_name_id = filename.split('_')[1].split('.')[0]\n        dat_type = dirname.split('/')[-1]\n        img_path = os.path.join(dirname, filename)\n        \n        kf.loc[kf.shape[0]] = [dat_type,std_id,filename,img_path]\n\n\npng_data = pd.DataFrame(kf.drop([0,1,2,3,4],axis=0).values,columns=kf.columns)\npng_data['image_file_ID'] = png_data.image_file_ID.str[13:-4]\npng_data\n\n# creating train and test datframes where all the image_study and studyInstanceUID are stored\npng_test = png_data[png_data['data_type'] == 'test']\npng_train = png_data[png_data['data_type'] == 'train']","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:30:40.609635Z","iopub.execute_input":"2021-07-24T04:30:40.609956Z","iopub.status.idle":"2021-07-24T04:31:13.704425Z","shell.execute_reply.started":"2021-07-24T04:30:40.609924Z","shell.execute_reply":"2021-07-24T04:31:13.702715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"png_train","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:31:13.707723Z","iopub.execute_input":"2021-07-24T04:31:13.708236Z","iopub.status.idle":"2021-07-24T04:31:13.73667Z","shell.execute_reply.started":"2021-07-24T04:31:13.708171Z","shell.execute_reply":"2021-07-24T04:31:13.735712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"png_test","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:31:35.105717Z","iopub.execute_input":"2021-07-24T04:31:35.106091Z","iopub.status.idle":"2021-07-24T04:31:35.12195Z","shell.execute_reply.started":"2021-07-24T04:31:35.106056Z","shell.execute_reply":"2021-07-24T04:31:35.120525Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv(\"/kaggle/input/siim-covid19-detection/sample_submission.csv\")\ntrain_image = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_image_level.csv\")\ntrain_study = pd.read_csv(\"/kaggle/input/siim-covid19-detection/train_study_level.csv\")\n\n# class label colum and case study id_number columns are created\n\ntrain_study['classification_class'] = np.zeros((train_study.shape[0],1))\nfor i in range(0,train_study.shape[0]):\n    train_study['classification_class'][i] = train_study.iloc[i].values[1:-1]\n\n# StudyInstanceUID \ntrain_study['StudyInstanceUID'] = train_study.id.str[0:-6]\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:32:04.944161Z","iopub.execute_input":"2021-07-24T04:32:04.944561Z","iopub.status.idle":"2021-07-24T04:32:09.053231Z","shell.execute_reply.started":"2021-07-24T04:32:04.944528Z","shell.execute_reply":"2021-07-24T04:32:09.052258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"itr = 0\ncl = []\nfor i in train_study['classification_class'].values:\n    if i[0]==1:\n        cl.append('Negative')\n    elif i[1]==1:\n        cl.append('Typical')\n    elif i[2]==1:\n        cl.append('Indeterminate')\n    elif i[3]==1:\n        cl.append('Atypical')\n    else:\n        a=1\n    itr = itr+1\n\ntrain_study['class_type'] = cl\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:32:17.505561Z","iopub.execute_input":"2021-07-24T04:32:17.506066Z","iopub.status.idle":"2021-07-24T04:32:17.559777Z","shell.execute_reply.started":"2021-07-24T04:32:17.506022Z","shell.execute_reply":"2021-07-24T04:32:17.558455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Selecting appropriate image for the StudyInstanceUID from coressponding multiple images_id's\n\nSIUID = train_image.groupby('StudyInstanceUID') #SIUID :- Syudy instance UID indexed dataframe\nim_fl_id = png_train.set_index('image_file_ID')\n\nImage_file_path = []\nImage_file_ID = []\nfor i in train_study['StudyInstanceUID'].values:\n    \n    pk = SIUID.get_group(i)\n    if (pk.shape[0] >1) & (pk.dropna(subset=['boxes']).shape[0]>0):\n        \n        im_id = SIUID.get_group(i).dropna(subset=['boxes']).values[0][0].split('_')[0]\n        path = im_fl_id.loc[im_id]['image_file_path']\n        Image_file_path.append(path)\n        Image_file_ID.append(im_id)\n    \n    else:\n        im_id = SIUID.get_group(i).values[0][0].split('_')[0]\n        path = im_fl_id.loc[im_id]['image_file_path']\n        Image_file_path.append(path)\n        Image_file_ID.append(im_id)\n\ntrain_study['image_file_path'] = Image_file_path\ntrain_study['image_file_ID'] = Image_file_ID\ntrain_study","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:32:28.25928Z","iopub.execute_input":"2021-07-24T04:32:28.259678Z","iopub.status.idle":"2021-07-24T04:32:40.612239Z","shell.execute_reply.started":"2021-07-24T04:32:28.259644Z","shell.execute_reply":"2021-07-24T04:32:40.611172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using this information for classification problem alone\ntrain_study.info()","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:32:48.873138Z","iopub.execute_input":"2021-07-24T04:32:48.873511Z","iopub.status.idle":"2021-07-24T04:32:48.895437Z","shell.execute_reply.started":"2021-07-24T04:32:48.873479Z","shell.execute_reply":"2021-07-24T04:32:48.894311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This data frame is for the object localisation and classification combined activities\n\ntrain_Study_image = pd.merge(train_image,train_study,on='StudyInstanceUID',how=\"inner\")\n#\ntrain_Study_image.columns = ['Image_id', 'boxes', 'label', 'StudyInstanceUID', 'Study_id',\n       'Negative for Pneumonia', 'Typical Appearance',\n       'Indeterminate Appearance', 'Atypical Appearance',\n       'classification_class','class_type','image_file_path','image_file_ID']\n\ntrain_Study_image","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:32:56.338796Z","iopub.execute_input":"2021-07-24T04:32:56.339554Z","iopub.status.idle":"2021-07-24T04:32:56.390052Z","shell.execute_reply.started":"2021-07-24T04:32:56.339506Z","shell.execute_reply":"2021-07-24T04:32:56.389242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Study Instance ID where all the images of the study ID are labelled as none\ntrain_Study_image.set_index('StudyInstanceUID').loc['0d962bb4c9f2']","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:33:06.583236Z","iopub.execute_input":"2021-07-24T04:33:06.583892Z","iopub.status.idle":"2021-07-24T04:33:06.607566Z","shell.execute_reply.started":"2021-07-24T04:33:06.583848Z","shell.execute_reply":"2021-07-24T04:33:06.606348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pre-Trained Model Activation ","metadata":{}},{"cell_type":"code","source":"img_size = 256\nimg_depth = 3\nmodel = tf.keras.applications.resnet50.ResNet50(include_top=False, #Do not include FC layer at the end\n                                          input_shape=(img_size,img_size, img_depth),\n                                          weights='imagenet')","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:33:19.991094Z","iopub.execute_input":"2021-07-24T04:33:19.991498Z","iopub.status.idle":"2021-07-24T04:33:26.166775Z","shell.execute_reply.started":"2021-07-24T04:33:19.991455Z","shell.execute_reply":"2021-07-24T04:33:26.165571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Set pre-trained model layers to not trainable\nfor layer in model.layers:\n    layer.trainable = False","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:33:35.870917Z","iopub.execute_input":"2021-07-24T04:33:35.871469Z","iopub.status.idle":"2021-07-24T04:33:35.881426Z","shell.execute_reply.started":"2021-07-24T04:33:35.871433Z","shell.execute_reply":"2021-07-24T04:33:35.880612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#get Output layer of Pre0trained model\nx = model.output\n\n#Flatten the output to feed to Dense layer\nx = tf.keras.layers.Flatten()(x)\n\n#Add one Dense layer\nx = tf.keras.layers.Dense(200, activation='relu')(x)\n\n#Add output layer\nprediction = tf.keras.layers.Dense(4,activation='softmax')(x)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:33:42.872902Z","iopub.execute_input":"2021-07-24T04:33:42.87347Z","iopub.status.idle":"2021-07-24T04:33:43.057279Z","shell.execute_reply.started":"2021-07-24T04:33:42.873431Z","shell.execute_reply":"2021-07-24T04:33:43.05588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Using Keras Model class\nfinal_model = tf.keras.models.Model(inputs=model.input, #Pre-trained model input as input layer\n                                    outputs=prediction) #Output layer added","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:33:49.429788Z","iopub.execute_input":"2021-07-24T04:33:49.430177Z","iopub.status.idle":"2021-07-24T04:33:49.449256Z","shell.execute_reply.started":"2021-07-24T04:33:49.430137Z","shell.execute_reply":"2021-07-24T04:33:49.448007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:33:52.619881Z","iopub.execute_input":"2021-07-24T04:33:52.62029Z","iopub.status.idle":"2021-07-24T04:33:52.645603Z","shell.execute_reply.started":"2021-07-24T04:33:52.620253Z","shell.execute_reply":"2021-07-24T04:33:52.644129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Image data augumentataion \ntrn_gen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255.,rotation_range=20,width_shift_range=0.2,height_shift_range=0.2,\n                                                shear_range=0.2,zoom_range=0.2,\n                                                fill_mode=\"nearest\",horizontal_flip=True,\n                                                #validation_split=0.2,\n                                                preprocessing_function=tf.keras.applications.resnet.preprocess_input)\n\n\nval_gen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255.,rotation_range=20,width_shift_range=0.2,height_shift_range=0.2,\n                                                shear_range=0.2,zoom_range=0.2,\n                                                fill_mode=\"nearest\",horizontal_flip=True,\n                                                #validation_split=0.2,\n                                                preprocessing_function=tf.keras.applications.resnet.preprocess_input)\n\ntest_gen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255.)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:33:59.54109Z","iopub.execute_input":"2021-07-24T04:33:59.541494Z","iopub.status.idle":"2021-07-24T04:33:59.549535Z","shell.execute_reply.started":"2021-07-24T04:33:59.541458Z","shell.execute_reply":"2021-07-24T04:33:59.54821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Just taking only the required columns from the trainstudy dataframe\nkj = train_study[['class_type','image_file_path']]\nkt = pd.get_dummies(train_study[['class_type']])\nfea = pd.concat([kt,kj],axis=1)\nfea.columns = ['Atypical', 'Indeterminate','Negative', 'Typical', 'class_type','image_file_path']\nfea","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:34:10.378604Z","iopub.execute_input":"2021-07-24T04:34:10.379134Z","iopub.status.idle":"2021-07-24T04:34:10.407877Z","shell.execute_reply.started":"2021-07-24T04:34:10.3791Z","shell.execute_reply":"2021-07-24T04:34:10.406793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_trn,X_val,Y_trn,Y_val = train_test_split(fea,fea[['class_type']],test_size=0.15,random_state=45)\n\n#Y_trn = tf.keras.utils.to_categorical(Y_trn, num_classes=4)\n#Y_val = tf.keras.utils.to_categorical(Y_val, num_classes=4)","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:35:00.981838Z","iopub.execute_input":"2021-07-24T04:35:00.982513Z","iopub.status.idle":"2021-07-24T04:35:01.203414Z","shell.execute_reply.started":"2021-07-24T04:35:00.982459Z","shell.execute_reply":"2021-07-24T04:35:01.202318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_trn\nX_val","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:35:07.312345Z","iopub.execute_input":"2021-07-24T04:35:07.312728Z","iopub.status.idle":"2021-07-24T04:35:07.329817Z","shell.execute_reply.started":"2021-07-24T04:35:07.312693Z","shell.execute_reply":"2021-07-24T04:35:07.328791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create train and test generator\nbatchsize = 128\ntrain_generator =  trn_gen.flow_from_dataframe (X_trn,\n                                                x_col=\"image_file_path\",\n                                                y_col=['Atypical', 'Indeterminate','Negative', 'Typical'],\n                                                target_size=(256, 256),\n                                                color_mode=\"rgb\",\n                                            #classes=['Negative','Typical','Indeterminate','Atypical'], \n                                                class_mode=\"raw\",\n                                                batch_size=batchsize,\n                                                shuffle=True,\n                                                seed=20,\n                                                ) #batchsize can be changed\n\nval_generator = val_gen.flow_from_dataframe (X_val,\n                                            x_col=\"image_file_path\",\n                                            y_col=['Atypical', 'Indeterminate','Negative', 'Typical'],\n                                            target_size=(256, 256),\n                                            color_mode=\"rgb\",\n                                            #classes=['Negative','Typical','Indeterminate','Atypical'], \n                                            class_mode=\"raw\",\n                                            batch_size=batchsize,\n                                            shuffle=True,\n                                            seed=20,\n                                            )","metadata":{"execution":{"iopub.status.busy":"2021-07-24T04:35:15.535637Z","iopub.execute_input":"2021-07-24T04:35:15.536014Z","iopub.status.idle":"2021-07-24T04:35:17.107577Z","shell.execute_reply.started":"2021-07-24T04:35:15.535982Z","shell.execute_reply":"2021-07-24T04:35:17.106729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model.fit_generator(train_generator, \n                          epochs=10,\n                          steps_per_epoch= X_trn.shape[0]//batchsize,\n                          validation_data=val_generator,\n                          validation_steps = X_val.shape[0]//batchsize,\n                          verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-06-22T10:17:09.258662Z","iopub.execute_input":"2021-06-22T10:17:09.259042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nprint(len(model.layers))\nfor layer in model.layers[150:]:\n    layer.trainable =  True \n'''","metadata":{"execution":{"iopub.status.busy":"2021-06-22T06:53:02.279185Z","iopub.status.idle":"2021-06-22T06:53:02.279808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}