{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Here Importing the Eseantial libraries **","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nimport os\nimport keras\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nimport cv2\nfrom keras import applications\nfrom keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, Input\nfrom keras.models import Model\nfrom keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:22:20.369414Z","iopub.execute_input":"2021-06-24T13:22:20.370036Z","iopub.status.idle":"2021-06-24T13:22:26.057516Z","shell.execute_reply.started":"2021-06-24T13:22:20.369945Z","shell.execute_reply":"2021-06-24T13:22:26.056500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the Lables of the images from train.csv file ","metadata":{}},{"cell_type":"code","source":"dataframe = pd.read_csv('/kaggle/input/human-protein-atlas-image-classification/train.csv')\ndataframe.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:22:33.636101Z","iopub.execute_input":"2021-06-24T13:22:33.636469Z","iopub.status.idle":"2021-06-24T13:22:33.705593Z","shell.execute_reply.started":"2021-06-24T13:22:33.636438Z","shell.execute_reply":"2021-06-24T13:22:33.704465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" * Defining the input shape of 1st input layer \n * decalare the batch size\n * save the path of train images in the variable 'path_to_train'","metadata":{}},{"cell_type":"code","source":"INPUT_SHAPE = (512, 512, 3)\nBATCH_SIZE = 16\npath_to_train = '/kaggle/input/human-protein-atlas-image-classification/train/'","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:22:36.901273Z","iopub.execute_input":"2021-06-24T13:22:36.901595Z","iopub.status.idle":"2021-06-24T13:22:36.905860Z","shell.execute_reply.started":"2021-06-24T13:22:36.901565Z","shell.execute_reply":"2021-06-24T13:22:36.904664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" Adding the column with the name of complete_path and load the full path of each image on csv ","metadata":{}},{"cell_type":"code","source":"dataframe[\"complete_path\"] = path_to_train + dataframe[\"Id\"]\ndataframe.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:22:39.501049Z","iopub.execute_input":"2021-06-24T13:22:39.501395Z","iopub.status.idle":"2021-06-24T13:22:39.532498Z","shell.execute_reply.started":"2021-06-24T13:22:39.501367Z","shell.execute_reply":"2021-06-24T13:22:39.531610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#  Visualizing the pictures\n\n","metadata":{}},{"cell_type":"code","source":"import random\nfig, axes = plt.subplots(3, 4, figsize=(10, 10))\nfor i in range(3):\n    for j in range(4):\n        idx = random.randint(0, dataframe.shape[0])\n        row = dataframe.iloc[idx,:]\n        path = row.complete_path\n        red = np.array(Image.open(path + '_red.png'))\n        green = np.array(Image.open(path + '_green.png'))\n        blue = np.array(Image.open(path + '_blue.png'))\n        im = np.stack((\n                red,\n                green,\n                blue),-1)\n        axes[i][j].imshow(im)\n        axes[i][j].set_title(row.Target)\n        axes[i][j].set_xticks([])\n        axes[i][j].set_yticks([])\nfig.tight_layout()\nfig.show(5);","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:22:42.948354Z","iopub.execute_input":"2021-06-24T13:22:42.948679Z","iopub.status.idle":"2021-06-24T13:22:44.591939Z","shell.execute_reply.started":"2021-06-24T13:22:42.948649Z","shell.execute_reply":"2021-06-24T13:22:44.591124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Spliting the data into parts train and val(valiation)","metadata":{}},{"cell_type":"code","source":"train, val = train_test_split(dataframe, test_size=0.2, random_state=42)\n","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:22:50.923534Z","iopub.execute_input":"2021-06-24T13:22:50.923884Z","iopub.status.idle":"2021-06-24T13:22:50.935771Z","shell.execute_reply.started":"2021-06-24T13:22:50.923833Z","shell.execute_reply":"2021-06-24T13:22:50.934774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Shape of train: {train.shape}')\nprint(f'Shape of val: {val.shape}')","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:22:54.717399Z","iopub.execute_input":"2021-06-24T13:22:54.717733Z","iopub.status.idle":"2021-06-24T13:22:54.722417Z","shell.execute_reply.started":"2021-06-24T13:22:54.717703Z","shell.execute_reply":"2021-06-24T13:22:54.721513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Cleaning the data for better results ","metadata":{}},{"cell_type":"code","source":"def get_clean_data(df):\n    targets = []\n    paths = []\n    for _, row in df.iterrows():\n        target_np = np.zeros((28))\n        t = [int(t) for t in row.Target.split()]\n        target_np[t] = 1\n        targets.append(target_np)\n        paths.append(row.complete_path)\n    return np.array(paths), np.array(targets)","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:22:57.063187Z","iopub.execute_input":"2021-06-24T13:22:57.063632Z","iopub.status.idle":"2021-06-24T13:22:57.075075Z","shell.execute_reply.started":"2021-06-24T13:22:57.063594Z","shell.execute_reply":"2021-06-24T13:22:57.073977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path, train_target = get_clean_data(train)\nval_path, val_target = get_clean_data(val)","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:22:59.789372Z","iopub.execute_input":"2021-06-24T13:22:59.789683Z","iopub.status.idle":"2021-06-24T13:23:02.550146Z","shell.execute_reply.started":"2021-06-24T13:22:59.789654Z","shell.execute_reply":"2021-06-24T13:23:02.549280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"printing train and val path and target\n\n","metadata":{}},{"cell_type":"code","source":"print(f'Train path shape: {train_path.shape}')\nprint(f'Train target shape: {train_target.shape}')\nprint(f'Val path shape: {val_path.shape}')\nprint(f'Val target shape: {val_target.shape}')\n","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:23:05.028891Z","iopub.execute_input":"2021-06-24T13:23:05.029231Z","iopub.status.idle":"2021-06-24T13:23:05.038752Z","shell.execute_reply.started":"2021-06-24T13:23:05.029198Z","shell.execute_reply":"2021-06-24T13:23:05.037793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"creating datasets from cleaned data\n","metadata":{}},{"cell_type":"code","source":"train_data = tf.data.Dataset.from_tensor_slices((train_path, train_target))\nval_data = tf.data.Dataset.from_tensor_slices((val_path, val_target))","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:23:07.869725Z","iopub.execute_input":"2021-06-24T13:23:07.870086Z","iopub.status.idle":"2021-06-24T13:23:09.667527Z","shell.execute_reply.started":"2021-06-24T13:23:07.870054Z","shell.execute_reply":"2021-06-24T13:23:09.666649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Experimental API for building input pipelines.\n\n","metadata":{}},{"cell_type":"code","source":"def load_data(path, target):\n    red = tf.squeeze(tf.image.decode_png(tf.io.read_file(path+'_red.png'), channels=1), [2])\n    blue = tf.squeeze(tf.image.decode_png(tf.io.read_file(path+'_blue.png'), channels=1), [2])\n    green = tf.squeeze(tf.image.decode_png(tf.io.read_file(path+'_green.png'), channels=1), [2])\n    #yellow=tf.squeeze(tf.image.decode_png(tf.io.read_file(path+'_yellow.png'), channels=1), [2])\n    img = tf.stack((\n                red,\n                green,\n                blue), axis=2)\n    return img, target\n\nAUTOTUNE = tf.data.experimental.AUTOTUNE\n\ntrain_data = train_data.map(load_data, num_parallel_calls=AUTOTUNE)\nval_data = val_data.map(load_data, num_parallel_calls=AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:23:12.967668Z","iopub.execute_input":"2021-06-24T13:23:12.968036Z","iopub.status.idle":"2021-06-24T13:23:13.116520Z","shell.execute_reply.started":"2021-06-24T13:23:12.968003Z","shell.execute_reply":"2021-06-24T13:23:13.115709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Adjust the contrast of an image or images by a random factor.\n\n","metadata":{}},{"cell_type":"code","source":"def image_augment(img, target):\n    img = tf.image.random_contrast(img, lower=0.3, upper=2.0)\n    img = tf.image.random_flip_up_down(img)\n    img = tf.image.random_brightness(img, max_delta=0.1)\n    return img, target\n    \ntrain_data = train_data.map(image_augment, num_parallel_calls=AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:23:16.044828Z","iopub.execute_input":"2021-06-24T13:23:16.045187Z","iopub.status.idle":"2021-06-24T13:23:16.149197Z","shell.execute_reply.started":"2021-06-24T13:23:16.045157Z","shell.execute_reply":"2021-06-24T13:23:16.148396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Better performance with the tf.data API\n","metadata":{}},{"cell_type":"code","source":"train_data_batches = train_data.batch(BATCH_SIZE).prefetch(buffer_size=AUTOTUNE)\nval_data_batches = val_data.batch(BATCH_SIZE).prefetch(buffer_size=AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:23:19.656426Z","iopub.execute_input":"2021-06-24T13:23:19.656816Z","iopub.status.idle":"2021-06-24T13:23:19.664874Z","shell.execute_reply.started":"2021-06-24T13:23:19.656779Z","shell.execute_reply":"2021-06-24T13:23:19.663661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Building the Model Architecture ","metadata":{}},{"cell_type":"markdown","source":"Using transfer learning and fine tuning on ReaNet50","metadata":{}},{"cell_type":"code","source":"resnet_model = applications.ResNet50(include_top=False, weights='imagenet')\n\nresnet_model.trainable = True\n\ninput_layer = Input(shape=INPUT_SHAPE)\nx = resnet_model(input_layer)\nx = Flatten()(x)\nx = Dropout(0.5)(x)\nx = Dense(512, activation='relu')(x)\nx = Dropout(0.5)(x)\noutput = Dense(28, activation='sigmoid')(x)\nmodel = Model(input_layer, output)\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:23:22.544422Z","iopub.execute_input":"2021-06-24T13:23:22.544737Z","iopub.status.idle":"2021-06-24T13:23:25.280085Z","shell.execute_reply.started":"2021-06-24T13:23:22.544706Z","shell.execute_reply":"2021-06-24T13:23:25.279296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"model compilation ","metadata":{}},{"cell_type":"code","source":"model.compile(optimizer=Adam(1e-3), loss='binary_crossentropy', metrics=['binary_accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:23:29.614657Z","iopub.execute_input":"2021-06-24T13:23:29.615061Z","iopub.status.idle":"2021-06-24T13:23:29.636372Z","shell.execute_reply.started":"2021-06-24T13:23:29.615026Z","shell.execute_reply":"2021-06-24T13:23:29.635371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now finally fit the model for trianing ...","metadata":{}},{"cell_type":"code","source":"history = model.fit(train_data_batches, steps_per_epoch = 150, validation_data = val_data_batches, epochs=10)","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:23:32.387264Z","iopub.execute_input":"2021-06-24T13:23:32.387598Z","iopub.status.idle":"2021-06-24T13:47:19.163034Z","shell.execute_reply.started":"2021-06-24T13:23:32.387569Z","shell.execute_reply":"2021-06-24T13:47:19.162136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualizing the Results ","metadata":{}},{"cell_type":"code","source":"binary_accuracy=history.history['binary_accuracy']\nval_binary_accuracy=history.history['val_binary_accuracy']\nepochs=range(1,len(binary_accuracy)+1)\nplt.plot(epochs,binary_accuracy,'b',label='Training accuracy')  \nplt.plot(epochs,val_binary_accuracy,'r',label='Validation accuracy')\nplt.title('Training and Validation accuracy')\nplt.legend()\nplt.figure()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:52:46.877603Z","iopub.execute_input":"2021-06-24T13:52:46.878006Z","iopub.status.idle":"2021-06-24T13:52:47.051484Z","shell.execute_reply.started":"2021-06-24T13:52:46.877962Z","shell.execute_reply":"2021-06-24T13:52:47.050637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss=history.history['loss']\nval_loss=history.history['val_loss']\n\nepochs=range(1,len(binary_accuracy)+1)\nplt.plot(epochs,loss,'b',label='Training loss')\nplt.plot(epochs,val_loss,'r',label='Validation loss')\nplt.title('Training and Validation loss')\nplt.legend()\nplt.figure()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:53:03.493844Z","iopub.execute_input":"2021-06-24T13:53:03.494204Z","iopub.status.idle":"2021-06-24T13:53:03.645134Z","shell.execute_reply.started":"2021-06-24T13:53:03.494173Z","shell.execute_reply":"2021-06-24T13:53:03.644281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('New_model.h5')","metadata":{"execution":{"iopub.status.busy":"2021-06-24T13:55:01.494983Z","iopub.execute_input":"2021-06-24T13:55:01.495313Z","iopub.status.idle":"2021-06-24T13:55:12.306720Z","shell.execute_reply.started":"2021-06-24T13:55:01.495284Z","shell.execute_reply":"2021-06-24T13:55:12.305763Z"},"trusted":true},"execution_count":null,"outputs":[]}]}