{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/working'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-06-10T01:45:59.085852Z","iopub.execute_input":"2021-06-10T01:45:59.086201Z","iopub.status.idle":"2021-06-10T01:45:59.683820Z","shell.execute_reply.started":"2021-06-10T01:45:59.086171Z","shell.execute_reply":"2021-06-10T01:45:59.676353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport matplotlib.pyplot as plt\nimport keras.backend as K\nfrom zipfile import ZipFile\nfrom IPython.display import clear_output\nfrom pathlib import Path\nimport zipfile","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:45:59.685262Z","iopub.execute_input":"2021-06-10T01:45:59.685558Z","iopub.status.idle":"2021-06-10T01:45:59.690378Z","shell.execute_reply.started":"2021-06-10T01:45:59.685531Z","shell.execute_reply":"2021-06-10T01:45:59.689361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data loading and preprocessing","metadata":{}},{"cell_type":"code","source":"train_zip_path = \"/kaggle/input/carvana-image-masking-challenge/train.zip\"\nwith zipfile.ZipFile(train_zip_path, \"r\") as z_:\n    z_.extractall(\"/kaggle/working\")","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:45:59.692185Z","iopub.execute_input":"2021-06-10T01:45:59.692510Z","iopub.status.idle":"2021-06-10T01:46:07.042070Z","shell.execute_reply.started":"2021-06-10T01:45:59.692482Z","shell.execute_reply":"2021-06-10T01:46:07.040894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"masks_zip_path = \"/kaggle/input/carvana-image-masking-challenge/train_masks.zip\"\nwith zipfile.ZipFile(masks_zip_path, \"r\") as z_:\n    z_.extractall(\"/kaggle/working\")\n","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:07.043705Z","iopub.execute_input":"2021-06-10T01:46:07.044034Z","iopub.status.idle":"2021-06-10T01:46:08.921615Z","shell.execute_reply.started":"2021-06-10T01:46:07.043998Z","shell.execute_reply":"2021-06-10T01:46:08.920522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(os.listdir(\"/kaggle/working/train\")))\nprint(len(os.listdir(\"/kaggle/working/train_masks\")))","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:08.922838Z","iopub.execute_input":"2021-06-10T01:46:08.923124Z","iopub.status.idle":"2021-06-10T01:46:08.934559Z","shell.execute_reply.started":"2021-06-10T01:46:08.923096Z","shell.execute_reply":"2021-06-10T01:46:08.933483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Train dataframe\n\ncar_ids = []\npaths = []\n\nfor dirname, _, filenames in os.walk(\"/kaggle/working/train\"):\n    for filename in filenames:\n        path = os.path.join(dirname, filename)\n        paths.append(path)\n        \n        car_id = filename.split(\".\")[0]\n        car_ids.append(car_id)\n        \ndf = pd.DataFrame({\"id\": car_ids, \"car_path\": paths})\ndf = df.set_index(\"id\")\ndf","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2021-06-10T01:46:08.935897Z","iopub.execute_input":"2021-06-10T01:46:08.936205Z","iopub.status.idle":"2021-06-10T01:46:08.997931Z","shell.execute_reply.started":"2021-06-10T01:46:08.936177Z","shell.execute_reply":"2021-06-10T01:46:08.996808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Train_mask dataframe\n\ncar_ids = []\nmask_path = []\n\nfor dirname, _,filenames in os.walk(\"/kaggle/working/train_masks\"):\n    for filename in filenames:\n        path = os.path.join(dirname, filename)\n        mask_path.append(path)\n        \n        car_id = filename.split(\".\")[0]\n        car_id = car_id.split(\"_mask\")[0]\n        car_ids.append(car_id)\n        \n        \nmask_df = pd.DataFrame({\"id\": car_ids, \"mask_path\": mask_path})\nmask_df = mask_df.set_index(\"id\")\nmask_df","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:08.999456Z","iopub.execute_input":"2021-06-10T01:46:08.999864Z","iopub.status.idle":"2021-06-10T01:46:09.040871Z","shell.execute_reply.started":"2021-06-10T01:46:08.999820Z","shell.execute_reply":"2021-06-10T01:46:09.039905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"mask_path\"] = mask_df[\"mask_path\"]\ndf = df.reset_index(drop=True)\ndf","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:09.043088Z","iopub.execute_input":"2021-06-10T01:46:09.043377Z","iopub.status.idle":"2021-06-10T01:46:09.061493Z","shell.execute_reply.started":"2021-06-10T01:46:09.043351Z","shell.execute_reply":"2021-06-10T01:46:09.060264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#data augmentation function\n\nimage_size = [256, 256]\nOUTPUT_CHANNELS = 3\n\ndef augmentation(input_image, mask_image):\n    \n    \n    if tf.random.uniform(()) > 0.5:\n        input_image = tf.image.flip_left_right(input_image)\n        mask_image = tf.image.flip_left_right(mask_image)\n    \n    return input_image, mask_image","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:09.064237Z","iopub.execute_input":"2021-06-10T01:46:09.064790Z","iopub.status.idle":"2021-06-10T01:46:09.074124Z","shell.execute_reply.started":"2021-06-10T01:46:09.064740Z","shell.execute_reply":"2021-06-10T01:46:09.072976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Preprocessing function\n\ndef preprocess(car_path, mask_path):\n    input_image = tf.io.read_file(car_path)\n    input_image = tf.image.decode_jpeg(input_image, channels=OUTPUT_CHANNELS)\n    input_image = tf.image.resize(input_image, image_size)\n    input_image = tf.cast(input_image, tf.float32) / 255.0\n\n    \n    mask_image = tf.io.read_file(mask_path)\n    mask_image = tf.image.decode_jpeg(mask_image, channels=OUTPUT_CHANNELS)\n    mask_image = tf.image.resize(mask_image, image_size)\n    mask_image = mask_image[:, :, :1]\n    mask_image = tf.math.sign(mask_image)\n    \n    input_image, mask_image = augmentation(input_image, mask_image)\n    \n    return input_image, mask_image","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:09.075547Z","iopub.execute_input":"2021-06-10T01:46:09.075838Z","iopub.status.idle":"2021-06-10T01:46:09.088844Z","shell.execute_reply.started":"2021-06-10T01:46:09.075810Z","shell.execute_reply":"2021-06-10T01:46:09.087984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#create_dataset function\n\ndef create_dataset(df, train = False):\n    if not train:\n        ds = tf.data.Dataset.from_tensor_slices((df[\"car_path\"].values, df[\"mask_path\"].values))\n        ds = ds.map(preprocess, num_parallel_calls=tf.data.AUTOTUNE)\n        \n    else:\n        ds = tf.data.Dataset.from_tensor_slices((df[\"car_path\"].values, df[\"mask_path\"].values))\n        ds = ds.map(preprocess, num_parallel_calls=tf.data.AUTOTUNE)\n        ds = ds.map(augmentation, num_parallel_calls=tf.data.AUTOTUNE)\n        \n    return ds","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:09.089862Z","iopub.execute_input":"2021-06-10T01:46:09.090255Z","iopub.status.idle":"2021-06-10T01:46:09.101855Z","shell.execute_reply.started":"2021-06-10T01:46:09.090214Z","shell.execute_reply":"2021-06-10T01:46:09.101084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Data split\n\nfrom sklearn.model_selection import train_test_split\n\ntrain_df, valid_df = train_test_split(df, random_state=42, test_size=0.25)\ntrain = create_dataset(train_df, train=True)\nvalid = create_dataset(valid_df)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:09.102836Z","iopub.execute_input":"2021-06-10T01:46:09.103234Z","iopub.status.idle":"2021-06-10T01:46:10.373581Z","shell.execute_reply.started":"2021-06-10T01:46:09.103206Z","shell.execute_reply":"2021-06-10T01:46:10.372549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_LENGTH = len(train_df)\nBATCH_SIZE = 16\nBUFFER_SIZE = 1000","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:10.375128Z","iopub.execute_input":"2021-06-10T01:46:10.375538Z","iopub.status.idle":"2021-06-10T01:46:10.380418Z","shell.execute_reply.started":"2021-06-10T01:46:10.375495Z","shell.execute_reply":"2021-06-10T01:46:10.379324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train and validation dataset\n\ntrain_dataset = train.cache().shuffle(BUFFER_SIZE).batch(BATCH_SIZE).repeat()\ntrain_dataset = train_dataset.prefetch(buffer_size=tf.data.AUTOTUNE)\nvalid_dataset = valid.batch(BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:10.382022Z","iopub.execute_input":"2021-06-10T01:46:10.382438Z","iopub.status.idle":"2021-06-10T01:46:10.406557Z","shell.execute_reply.started":"2021-06-10T01:46:10.382397Z","shell.execute_reply":"2021-06-10T01:46:10.405517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Take a look before data training\n\ndef display(display_list):\n    plt.figure(figsize=(15,15))\n    \n    title = [\"Input image\", \"True mask\", \"Predicted_mask\"]\n    \n    for i in range(len(display_list)):\n        plt.subplot(1, len(display_list), i + 1)\n        plt.title(title[i])           \n        plt.imshow(tf.keras.preprocessing.image.array_to_img(display_list[i]))\n        plt.axis(\"off\")\n                   \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:10.408385Z","iopub.execute_input":"2021-06-10T01:46:10.408780Z","iopub.status.idle":"2021-06-10T01:46:10.415227Z","shell.execute_reply.started":"2021-06-10T01:46:10.408745Z","shell.execute_reply":"2021-06-10T01:46:10.414095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for image, mask in train.take(1):\n    sample_image, sample_mask = image, mask\n    display([sample_image, sample_mask])","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:10.417089Z","iopub.execute_input":"2021-06-10T01:46:10.417569Z","iopub.status.idle":"2021-06-10T01:46:10.958585Z","shell.execute_reply.started":"2021-06-10T01:46:10.417483Z","shell.execute_reply":"2021-06-10T01:46:10.957356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model building","metadata":{}},{"cell_type":"code","source":"#Base model\n\nbase_model = tf.keras.applications.MobileNetV2(input_shape=[256, 256, 3], include_top=False)\n\nlayer_names = [\n    \"block_1_expand_relu\",\n    \"block_3_expand_relu\",\n    \"block_6_expand_relu\",\n    \"block_13_expand_relu\",\n    \"block_16_project\",\n]","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:10.960286Z","iopub.execute_input":"2021-06-10T01:46:10.960620Z","iopub.status.idle":"2021-06-10T01:46:12.312817Z","shell.execute_reply.started":"2021-06-10T01:46:10.960589Z","shell.execute_reply":"2021-06-10T01:46:12.311624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Encoder\n\nmodel_base_output = [base_model.get_layer(name).output for name in layer_names]\n\ndown_stack = tf.keras.Model(inputs=base_model.input, outputs=model_base_output)\n\ndown_stack.trainable = False","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:12.313968Z","iopub.execute_input":"2021-06-10T01:46:12.314269Z","iopub.status.idle":"2021-06-10T01:46:12.341535Z","shell.execute_reply.started":"2021-06-10T01:46:12.314224Z","shell.execute_reply":"2021-06-10T01:46:12.340306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Decoder function\n\ndef upsample(filters, size, apply_dropout=False):\n    initializer = tf.random_normal_initializer(0., 0.02)\n    \n    result = tf.keras.Sequential()\n    result.add(\n    tf.keras.layers.Conv2DTranspose(filters, size, strides=2, padding=\"same\", kernel_initializer=initializer, use_bias=False))\n    \n    result.add(tf.keras.layers.BatchNormalization())\n    \n    if apply_dropout:\n        result.add(tf.keras.layers.Dropout(0.5))\n        \n    result.add(tf.keras.layers.ReLU())\n\n    return result","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:12.343013Z","iopub.execute_input":"2021-06-10T01:46:12.343370Z","iopub.status.idle":"2021-06-10T01:46:12.349880Z","shell.execute_reply.started":"2021-06-10T01:46:12.343339Z","shell.execute_reply":"2021-06-10T01:46:12.349151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Decoder\n\nup_stack = [\n    upsample(512, 3),\n    upsample(256, 3),\n    upsample(128, 3),\n    upsample(64, 3)\n] ","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:12.351148Z","iopub.execute_input":"2021-06-10T01:46:12.351727Z","iopub.status.idle":"2021-06-10T01:46:12.394691Z","shell.execute_reply.started":"2021-06-10T01:46:12.351690Z","shell.execute_reply":"2021-06-10T01:46:12.393672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#unet_model function\n\ndef unet_model(output_channels):\n    inputs = tf.keras.layers.Input(shape=[256, 256, 3])\n    \n    skips = down_stack(inputs)\n    x = skips[-1]\n    skips = reversed(skips[:-1])\n    \n    for up, skip, in zip(up_stack, skips):\n        x = up(x)\n        concat = tf.keras.layers.Concatenate()\n        x = concat([x, skip])\n        \n    last = tf.keras.layers.Conv2DTranspose(output_channels, 3, strides=2, padding=\"same\")\n    x = last(x)\n    \n    return tf.keras.Model(inputs=inputs, outputs=x)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:12.395746Z","iopub.execute_input":"2021-06-10T01:46:12.396154Z","iopub.status.idle":"2021-06-10T01:46:12.402926Z","shell.execute_reply.started":"2021-06-10T01:46:12.396124Z","shell.execute_reply":"2021-06-10T01:46:12.402057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Create model\n\nmodel = unet_model(OUTPUT_CHANNELS)\n\nmodel.compile(optimizer=\"adam\",\n             loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n             metrics=[\"accuracy\"])","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:12.403973Z","iopub.execute_input":"2021-06-10T01:46:12.404450Z","iopub.status.idle":"2021-06-10T01:46:13.018739Z","shell.execute_reply.started":"2021-06-10T01:46:12.404419Z","shell.execute_reply":"2021-06-10T01:46:13.017892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Model architecture\n\ntf.keras.utils.plot_model(model, show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:13.021333Z","iopub.execute_input":"2021-06-10T01:46:13.021790Z","iopub.status.idle":"2021-06-10T01:46:13.530116Z","shell.execute_reply.started":"2021-06-10T01:46:13.021760Z","shell.execute_reply":"2021-06-10T01:46:13.529028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Let's test the model to see what it predicts before training\n\ndef create_mask(pred_mask):\n    pred_mask = tf.argmax(pred_mask, axis=-1)\n    pred_mask = pred_mask[..., tf.newaxis]\n    return pred_mask[0]","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:13.531890Z","iopub.execute_input":"2021-06-10T01:46:13.532175Z","iopub.status.idle":"2021-06-10T01:46:13.537369Z","shell.execute_reply.started":"2021-06-10T01:46:13.532146Z","shell.execute_reply":"2021-06-10T01:46:13.536347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_predictions(train_dataset=None, num=1):\n    if train_dataset:\n        for image, mask in train_dataset.take(num):\n            pred_mask = model.predict(image)\n            display([image[0], mask[0], create_mask(pred_mask)])\n            \n    else:\n        display([sample_image, sample_mask, create_mask(model.predict(sample_image[tf.newaxis, ...]))])\n            \nshow_predictions()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:13.539002Z","iopub.execute_input":"2021-06-10T01:46:13.539300Z","iopub.status.idle":"2021-06-10T01:46:15.128220Z","shell.execute_reply.started":"2021-06-10T01:46:13.539271Z","shell.execute_reply":"2021-06-10T01:46:15.127220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:15.129505Z","iopub.execute_input":"2021-06-10T01:46:15.129811Z","iopub.status.idle":"2021-06-10T01:46:15.155513Z","shell.execute_reply.started":"2021-06-10T01:46:15.129780Z","shell.execute_reply":"2021-06-10T01:46:15.154532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Calllback function\n\nfrom IPython.display import clear_output\n\nclass DisplayCallback(tf.keras.callbacks.Callback):\n    def on_epoch_begin(self, epoch, logs=None):\n        clear_output(wait=True)\n        show_predictions()\n        print(\"\\nSample Predictions after epoch {}\\n\".format(epoch+1))","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:15.156765Z","iopub.execute_input":"2021-06-10T01:46:15.157063Z","iopub.status.idle":"2021-06-10T01:46:15.162563Z","shell.execute_reply.started":"2021-06-10T01:46:15.157031Z","shell.execute_reply":"2021-06-10T01:46:15.161450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"EPOCHS = 5\nSTEPS_PER_EPOCH= TRAIN_LENGTH // BATCH_SIZE\n\nmodel_history = model.fit(train_dataset, epochs=EPOCHS,\n                          steps_per_epoch=STEPS_PER_EPOCH,\n                          validation_data=valid_dataset,\n                          callbacks=[DisplayCallback()])","metadata":{"execution":{"iopub.status.busy":"2021-06-10T01:46:15.164141Z","iopub.execute_input":"2021-06-10T01:46:15.164524Z","iopub.status.idle":"2021-06-10T02:56:06.978769Z","shell.execute_reply.started":"2021-06-10T01:46:15.164474Z","shell.execute_reply":"2021-06-10T02:56:06.977395Z"},"trusted":true},"execution_count":null,"outputs":[]}]}