{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport pandas as pd\nimport os\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras import backend as K\nimport keras\nfrom keras.models import Sequential, Model,load_model\nfrom tensorflow.keras.utils import Sequence\n# from keras.optimizers import SGD\nfrom tensorflow.keras.optimizers import SGD\nfrom keras.callbacks import EarlyStopping,ModelCheckpoint\n# from google.colab.patches import cv2_imshow\nfrom keras.layers import Input, Add, Dense, Activation, ZeroPadding2D, BatchNormalization, Flatten, Conv2D, AveragePooling2D, MaxPooling2D, GlobalMaxPooling2D,MaxPool2D\nfrom keras.preprocessing import image\nfrom keras.initializers import glorot_uniform\nfrom pathlib import Path\nimport tensorflow as tf\nfrom tensorflow.keras.layers import (Dense, Dropout, Activation, Flatten, Input, Add,\n                                    BatchNormalization, LeakyReLU, Concatenate, GlobalAveragePooling2D,Conv2D, AveragePooling2D)\nfrom sklearn.model_selection import train_test_split\nimport pydicom\nimport matplotlib as plt","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:41.842604Z","iopub.execute_input":"2022-07-23T20:24:41.843655Z","iopub.status.idle":"2022-07-23T20:24:41.852378Z","shell.execute_reply.started":"2022-07-23T20:24:41.843612Z","shell.execute_reply":"2022-07-23T20:24:41.851109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROOT = Path('../input/osic-pulmonary-fibrosis-progression/')\n\ntrain = pd.read_csv(ROOT / 'train.csv')\ntest = pd.read_csv(ROOT / 'test.csv')\nsub = pd.read_csv(ROOT / 'sample_submission.csv')\n\nsub.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:42.185397Z","iopub.execute_input":"2022-07-23T20:24:42.186018Z","iopub.status.idle":"2022-07-23T20:24:42.246011Z","shell.execute_reply.started":"2022-07-23T20:24:42.185984Z","shell.execute_reply":"2022-07-23T20:24:42.244919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_tab(df):\n    ''' \n    This function gives an array wrt each patient containing\n    feature like age, gender and smoking status\n    '''\n    vector = [(df.Age.values[0]-30)/30]\n    \n    if df.Sex.values[0].lower() == 'male':\n        vector.append(0)\n    else:\n        vector.append(1)\n        \n    if df.SmokingStatus.values[0] == 'Never smoked':\n        vector.extend([0,0])\n    elif df.SmokingStatus.values[0] == 'Ex-smoker':\n        vector.extend([1,1])\n    elif df.SmokingStatus.values[0] == 'Currently smokes':\n        vector.extend([0,1])\n    else:\n        vector.extend([1,0])\n        \n    return np.array(vector)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:42.452721Z","iopub.execute_input":"2022-07-23T20:24:42.453305Z","iopub.status.idle":"2022-07-23T20:24:42.462267Z","shell.execute_reply.started":"2022-07-23T20:24:42.453251Z","shell.execute_reply":"2022-07-23T20:24:42.461316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_img(path):\n    d = pydicom.dcmread(path)\n    img = cv2.resize(d.pixel_array/2**11, (224,224))\n    stacked_img = np.stack((img,)*3, axis=-1)\n    return stacked_img","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:42.656977Z","iopub.execute_input":"2022-07-23T20:24:42.657873Z","iopub.status.idle":"2022-07-23T20:24:42.663943Z","shell.execute_reply.started":"2022-07-23T20:24:42.657832Z","shell.execute_reply":"2022-07-23T20:24:42.662979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get_img('../input/osic-pulmonary-fibrosis-progression/train/ID00007637202177411956430/1.dcm')","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:42.905236Z","iopub.execute_input":"2022-07-23T20:24:42.905924Z","iopub.status.idle":"2022-07-23T20:24:42.910672Z","shell.execute_reply.started":"2022-07-23T20:24:42.905884Z","shell.execute_reply":"2022-07-23T20:24:42.909614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"A = {} #Stores slope value for each of the patient\nTAB = {} #Stores training data wrt each patient\nP = [] #Stores all unique patient id's\n\nfor i,p in enumerate(train.Patient.unique()):\n    sub = train.loc[train.Patient == p, :]\n    fvc = sub.FVC.values\n    week = sub.Weeks.values\n    #print(week)\n    c = np.vstack([week, np.ones(len(week))]).T\n    a, b = np.linalg.lstsq(c,fvc)[0]\n    #print(b)\n    \n    A[p] = a # Contains slope\n    TAB[p] = get_tab(sub) #Contains gender and smoking feature\n    P.append(p) #contains unique id\n   # print(TAB)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:43.127935Z","iopub.execute_input":"2022-07-23T20:24:43.128468Z","iopub.status.idle":"2022-07-23T20:24:43.343249Z","shell.execute_reply.started":"2022-07-23T20:24:43.128439Z","shell.execute_reply":"2022-07-23T20:24:43.341447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class IGenerator(Sequence):\n    \n    ''' \n    This is the generator class, which generates an input of batch size 32\n    i.e 32 patient's 2 dicom image, and features from tabular data is generated. As output \n    from his generator x and y contains pixel_data of a dicom image, tab conatins patient's meta\n    information, and 'a' is the coeffiecient wrt each patient. \n    '''\n    BAD_ID = ['ID00011637202177653955184', 'ID00052637202186188008618']\n    def __init__(self, keys, a, tab, batch_size=16):\n        self.keys = [k for k in keys if k not in self.BAD_ID]\n        self.a = a\n        self.tab = tab\n        self.batch_size = batch_size\n        \n        self.train_data = {}\n        for p in train.Patient.values:\n            self.train_data[p] = os.listdir(f'../input/osic-pulmonary-fibrosis-progression/train/{p}/')\n            #print(p)\n    def __len__(self):\n        return 1000\n    \n    def __getitem__(self, idx):\n        x = []\n        a, tab = [], [] \n        keys = np.random.choice(self.keys, size = self.batch_size)\n        \n        for k in keys:\n            try:\n                i = np.random.choice(self.train_data[k], size=1)[0]\n                img1 = get_img(f'../input/osic-pulmonary-fibrosis-progression/train/{k}/{i}')\n                x.append(img1)\n                a.append(self.a[k])\n                tab.append(self.tab[k])\n            except:\n                print(k, i)     \n        \n        x,a,tab = np.array(x), np.array(a), np.array(tab)\n#         x = np.expand_dims(x, axis=-1)\n        \n        return [x,tab] , a","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:43.378701Z","iopub.execute_input":"2022-07-23T20:24:43.379224Z","iopub.status.idle":"2022-07-23T20:24:43.391373Z","shell.execute_reply.started":"2022-07-23T20:24:43.379185Z","shell.execute_reply":"2022-07-23T20:24:43.390375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Implementation of Identity Block","metadata":{}},{"cell_type":"code","source":"def identity_block(X, f, filters, stage, block):\n   \n    conv_name_base = 'res' + str(stage) + block + '_branch'\n    bn_name_base = 'bn' + str(stage) + block + '_branch'\n    F1, F2, F3 = filters\n\n    X_shortcut = X\n   \n    X = Conv2D(filters=F1, kernel_size=(1, 1), strides=(1, 1), padding='valid', name=conv_name_base + '2a', kernel_initializer=glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis=3, name=bn_name_base + '2a')(X)\n    X = Activation('relu')(X)\n\n    X = Conv2D(filters=F2, kernel_size=(f, f), strides=(1, 1), padding='same', name=conv_name_base + '2b', kernel_initializer=glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis=3, name=bn_name_base + '2b')(X)\n    X = Activation('relu')(X)\n\n    X = Conv2D(filters=F3, kernel_size=(1, 1), strides=(1, 1), padding='valid', name=conv_name_base + '2c', kernel_initializer=glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis=3, name=bn_name_base + '2c')(X)\n\n    X = Add()([X, X_shortcut])# SKIP Connection\n    X = Activation('relu')(X)\n\n    return X","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:43.850155Z","iopub.execute_input":"2022-07-23T20:24:43.85045Z","iopub.status.idle":"2022-07-23T20:24:43.860707Z","shell.execute_reply.started":"2022-07-23T20:24:43.850424Z","shell.execute_reply":"2022-07-23T20:24:43.859613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Implementation of Convolutional Block","metadata":{}},{"cell_type":"code","source":"def convolutional_block(X, f, filters, stage, block, s=2):\n   \n    conv_name_base = 'res' + str(stage) + block + '_branch'\n    bn_name_base = 'bn' + str(stage) + block + '_branch'\n\n    F1, F2, F3 = filters\n\n    X_shortcut = X\n\n    X = Conv2D(filters=F1, kernel_size=(1, 1), strides=(s, s), padding='valid', name=conv_name_base + '2a', kernel_initializer=glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis=3, name=bn_name_base + '2a')(X)\n    X = Activation('relu')(X)\n\n    X = Conv2D(filters=F2, kernel_size=(f, f), strides=(1, 1), padding='same', name=conv_name_base + '2b', kernel_initializer=glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis=3, name=bn_name_base + '2b')(X)\n    X = Activation('relu')(X)\n\n    X = Conv2D(filters=F3, kernel_size=(1, 1), strides=(1, 1), padding='valid', name=conv_name_base + '2c', kernel_initializer=glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis=3, name=bn_name_base + '2c')(X)\n\n    X_shortcut = Conv2D(filters=F3, kernel_size=(1, 1), strides=(s, s), padding='valid', name=conv_name_base + '1', kernel_initializer=glorot_uniform(seed=0))(X_shortcut)\n    X_shortcut = BatchNormalization(axis=3, name=bn_name_base + '1')(X_shortcut)\n\n    X = Add()([X, X_shortcut])\n    X = Activation('relu')(X)\n\n    return X","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:44.330386Z","iopub.execute_input":"2022-07-23T20:24:44.330722Z","iopub.status.idle":"2022-07-23T20:24:44.341912Z","shell.execute_reply.started":"2022-07-23T20:24:44.330689Z","shell.execute_reply":"2022-07-23T20:24:44.340796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Implementation of ResNet-50","metadata":{}},{"cell_type":"code","source":"def ResNet50(input_shape=(224, 224, 3)):\n\n    X_input = Input(input_shape)\n\n    X = ZeroPadding2D((3, 3))(X_input)\n\n    X = Conv2D(64, (7, 7), strides=(2, 2), name='conv1', kernel_initializer=glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis=3, name='bn_conv1')(X)\n    X = Activation('relu')(X)\n    X = MaxPooling2D((3, 3), strides=(2, 2))(X)\n\n    X = convolutional_block(X, f=3, filters=[64, 64, 256], stage=2, block='a', s=1)\n    X = identity_block(X, 3, [64, 64, 256], stage=2, block='b')\n    X = identity_block(X, 3, [64, 64, 256], stage=2, block='c')\n\n\n    X = convolutional_block(X, f=3, filters=[128, 128, 512], stage=3, block='a', s=2)\n    X = identity_block(X, 3, [128, 128, 512], stage=3, block='b')\n    X = identity_block(X, 3, [128, 128, 512], stage=3, block='c')\n    X = identity_block(X, 3, [128, 128, 512], stage=3, block='d')\n\n    X = convolutional_block(X, f=3, filters=[256, 256, 1024], stage=4, block='a', s=2)\n    X = identity_block(X, 3, [256, 256, 1024], stage=4, block='b')\n    X = identity_block(X, 3, [256, 256, 1024], stage=4, block='c')\n    X = identity_block(X, 3, [256, 256, 1024], stage=4, block='d')\n    X = identity_block(X, 3, [256, 256, 1024], stage=4, block='e')\n    X = identity_block(X, 3, [256, 256, 1024], stage=4, block='f')\n\n    X = X = convolutional_block(X, f=3, filters=[512, 512, 2048], stage=5, block='a', s=2)\n    X = identity_block(X, 3, [512, 512, 2048], stage=5, block='b')\n    X = identity_block(X, 3, [512, 512, 2048], stage=5, block='c')\n\n    X = AveragePooling2D(pool_size=(2, 2), padding='same')(X)\n    \n    model = Model(inputs=X_input, outputs=X, name='ResNet50')\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:44.82139Z","iopub.execute_input":"2022-07-23T20:24:44.821725Z","iopub.status.idle":"2022-07-23T20:24:44.837778Z","shell.execute_reply.started":"2022-07-23T20:24:44.821691Z","shell.execute_reply":"2022-07-23T20:24:44.836652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = ResNet50(input_shape=(224, 224, 3))","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:45.028268Z","iopub.execute_input":"2022-07-23T20:24:45.028724Z","iopub.status.idle":"2022-07-23T20:24:48.762082Z","shell.execute_reply.started":"2022-07-23T20:24:45.02869Z","shell.execute_reply":"2022-07-23T20:24:48.761122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# input3 = Input(shape=(4,))\n#    z = tf.keras.layers.GaussianNoise(0.2)(input3)\n#    xyz = Concatenate()([l16, z])\n#    xyz = Dropout(0.6)(xyz) \n#    xyz = Dense(1)(xyz)\n#    return Model([input_img, input3] , xyz)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:48.764121Z","iopub.execute_input":"2022-07-23T20:24:48.76469Z","iopub.status.idle":"2022-07-23T20:24:48.768936Z","shell.execute_reply.started":"2022-07-23T20:24:48.764638Z","shell.execute_reply":"2022-07-23T20:24:48.767939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"headModel = base_model.output\nheadModel = Flatten()(headModel)\nheadModel=Dense(256, activation='relu', name='fc1',kernel_initializer=glorot_uniform(seed=0))(headModel)\nheadModel=Dense(128, activation='relu', name='fc2',kernel_initializer=glorot_uniform(seed=0))(headModel)\ninput3 = Input(shape=(4,))\nz = tf.keras.layers.GaussianNoise(0.2)(input3)\nxyz = Concatenate()([headModel, z])\nxyz = Dropout(0.6)(xyz) \nxyz = Dense(1)(xyz)\n# headModel = Dense( 1,activation='sigmoid', name='fc3',kernel_initializer=glorot_uniform(seed=0))(headModel)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:48.770467Z","iopub.execute_input":"2022-07-23T20:24:48.77107Z","iopub.status.idle":"2022-07-23T20:24:48.809443Z","shell.execute_reply.started":"2022-07-23T20:24:48.771034Z","shell.execute_reply":"2022-07-23T20:24:48.808619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Model(inputs=[base_model.input,input3], outputs=xyz)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:48.811846Z","iopub.execute_input":"2022-07-23T20:24:48.812167Z","iopub.status.idle":"2022-07-23T20:24:48.830373Z","shell.execute_reply.started":"2022-07-23T20:24:48.812135Z","shell.execute_reply":"2022-07-23T20:24:48.82955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:48.833252Z","iopub.execute_input":"2022-07-23T20:24:48.834099Z","iopub.status.idle":"2022-07-23T20:24:48.862007Z","shell.execute_reply.started":"2022-07-23T20:24:48.834066Z","shell.execute_reply":"2022-07-23T20:24:48.861144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from tensorflow import keras\ntf.keras.utils.plot_model(model,'img.png', show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:48.863159Z","iopub.execute_input":"2022-07-23T20:24:48.863641Z","iopub.status.idle":"2022-07-23T20:24:51.396382Z","shell.execute_reply.started":"2022-07-23T20:24:48.863609Z","shell.execute_reply":"2022-07-23T20:24:51.393243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=tf.keras.optimizers.SGD(learning_rate=0.0001), loss=\"mae\")","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:51.398005Z","iopub.execute_input":"2022-07-23T20:24:51.398381Z","iopub.status.idle":"2022-07-23T20:24:51.425305Z","shell.execute_reply.started":"2022-07-23T20:24:51.398346Z","shell.execute_reply":"2022-07-23T20:24:51.424278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tr_p, vl_p = train_test_split(P, shuffle=True, train_size= 0.8)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:51.426906Z","iopub.execute_input":"2022-07-23T20:24:51.427485Z","iopub.status.idle":"2022-07-23T20:24:51.432712Z","shell.execute_reply.started":"2022-07-23T20:24:51.427451Z","shell.execute_reply":"2022-07-23T20:24:51.431877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.config.run_functions_eagerly(True)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:51.434351Z","iopub.execute_input":"2022-07-23T20:24:51.435074Z","iopub.status.idle":"2022-07-23T20:24:51.441506Z","shell.execute_reply.started":"2022-07-23T20:24:51.43504Z","shell.execute_reply":"2022-07-23T20:24:51.440454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hist1 = model.fit_generator(IGenerator(keys=tr_p, \n                               a = A, \n                               tab = TAB), \n                    steps_per_epoch = 200,\n                    validation_data=IGenerator(keys=vl_p, \n                               a = A, \n                               tab = TAB),\n                    validation_steps = 20,\n                    epochs=500)","metadata":{"execution":{"iopub.status.busy":"2022-07-23T20:24:51.445104Z","iopub.execute_input":"2022-07-23T20:24:51.445723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, psutil  \n\ndef cpu_stats():\n    pid = os.getpid()\n    py = psutil.Process(pid)\n    memory_use = py.memory_info()[0] / 2. ** 30\n    return 'memory GB:' + str(np.round(memory_use, 2))\ncpu_stats()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(hist1.history['loss'])\nplt.plot(hist1.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'validation'], loc='upper left')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}