{"cells":[{"metadata":{},"cell_type":"markdown","source":"## Predicting Stable Structures\n\nThis is starter notebook for tensorflow users for this competition. It uses tf.data for loading data for training and model is trained using transfer learning."},{"metadata":{},"cell_type":"markdown","source":"Importing required dependencies"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport os\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Getting Path to csv files:\n\n- train.csv\n- test.csv \n- submission.csv"},{"metadata":{"trusted":true},"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        if filename.endswith(\".csv\"):\n            print(os.path.join(dirname, filename))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"base_path=\"/kaggle/input/applications-of-deep-learning-wustl-fall-2020/final-kaggle-data\"\nsubmission_file=\"/kaggle/input/applications-of-deep-learning-wustl-fall-2020/final-kaggle-data/submit.csv\"\ntest_path=\"/kaggle/input/applications-of-deep-learning-wustl-fall-2020/final-kaggle-data/test.csv\"\ntrain_path=\"/kaggle/input/applications-of-deep-learning-wustl-fall-2020/final-kaggle-data/train.csv\"","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Reading CSV files and viewing their heads"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_data=pd.read_csv(train_path)\ntest_data=pd.read_csv(test_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_data.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission=pd.read_csv(submission_file)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Remove curroupted files\n\n- remove_defected_images : remove corrupted images (if previously known) from data\n- check_and_remove_defected_images : check and remove any corrupted image"},{"metadata":{"trusted":true},"cell_type":"code","source":"def remove_defected_images(ids,labels,to_remove):\n    defected = []\n    for i,img_id in enumerate(ids):\n        if img_id in to_remove:\n            defected.append(img_id)\n            ids = np.delete(ids,i)\n            labels = np.delete(labels,i)\n    return defected,ids,labels","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def check_and_remove_defected_images(ids,labels):\n    defected = []\n    for i,img_id in enumerate(ids):\n        try:\n            image = tf.io.read_file(get_path_of_image(img_id))\n            image = tf.image.decode_png(image,channels=3)\n        except:\n            defected.append(img_id)\n            ids = np.delete(ids,i)\n            labels = np.delete(labels,i)\n    return defected,ids,labels","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"remove any corrupted png files from dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"defected,dataX,dataY = remove_defected_images(train_data.iloc[:,0].values,\n                                                train_data.iloc[:,1].values,\n                                                [1300])\nprint(\"Train Elements Defected: \",defected)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Splitting data into training and evaluation"},{"metadata":{"trusted":true},"cell_type":"code","source":"trainX,evalX,trainY,evalY = train_test_split(dataX,\n                                             dataY,\n                                             random_state=11,\n                                             test_size=0.1\n                                            )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"testX = test_data.iloc[:,0].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"num_train_images = len(trainX)\nnum_eval_images=len(evalX)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Number of train images: \",num_train_images)\nprint(\"Number of eval images: \",num_eval_images)\nprint(\"Number of test images: \",len(testX))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Setting Parameters here"},{"metadata":{"trusted":true},"cell_type":"code","source":"EPOCHS=50\nBATCH_SIZE=32\nIMAGE_DIM=(192,192)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Helper Functions\n\n- get_path_of_image : for getting path to image from its id\n- load_tf_image : loading and normalizing image and converting to tensor\n- generate_tf_dataset : generate tf dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_path_of_image(image_id):\n    return os.path.join(base_path,f\"{image_id}.png\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def load_tf_image(image_path,dim):\n    image = tf.io.read_file(image_path)\n    image = tf.image.decode_png(image,channels=3)\n    image = tf.image.resize(image,dim)\n    image = tf.image.convert_image_dtype(image, tf.float32)\n    image = image/255.0\n    return image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def generate_tf_dataset(X,Y,image_size):\n    X = [get_path_of_image(str(x)) for x in X]\n    datasetX = tf.data.Dataset.from_tensor_slices(X).map(\n            lambda path: load_tf_image(path,image_size),\n            num_parallel_calls=tf.data.experimental.AUTOTUNE\n    )\n    datasetY = tf.data.Dataset.from_tensor_slices(Y)\n    dataset = tf.data.Dataset.zip((datasetX,datasetY))\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.repeat()\n    dataset = dataset.prefetch(buffer_size=tf.data.experimental.AUTOTUNE)\n    return dataset","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def plot_images_grid(data,num_rows=1,labels=None,class_names=None):\n    images, labels = data\n    n=len(images)\n    if n > 1:\n        num_cols=np.ceil(n/num_rows)\n        fig,axes=plt.subplots(ncols=int(num_cols),nrows=int(num_rows))\n        axes=axes.flatten()\n        fig.set_size_inches((20,20))\n        for i,image in enumerate(images):\n            axes[i].axis('off')\n            axes[i].imshow(image.numpy())\n            label = labels[i].numpy()\n            axes[i].set_title(class_names[label])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Setting up train and eval tf dataset\n\nCreate tf datasets using tf.data for training and validation. A single element of these datasets return *(Image,Label)* where Image = *(batch_size,image_width,image_height,channels)* and Label = *(batch_size,)*. "},{"metadata":{"trusted":true},"cell_type":"code","source":"train_dataset=generate_tf_dataset(trainX,trainY,IMAGE_DIM)\nprint(train_dataset.element_spec)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"eval_dataset=generate_tf_dataset(evalX,evalY,IMAGE_DIM)\nprint(eval_dataset.element_spec)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Plotting and Visualizing images"},{"metadata":{"trusted":true},"cell_type":"code","source":"class_names=[\"not stable\",\"stable\"]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_images_grid(next(iter(train_dataset.take(1))),class_names=class_names,num_rows=4)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plot_images_grid(next(iter(eval_dataset.take(1))),class_names=class_names,num_rows=4)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Training Model\n\nIt uses densenet201 pretrained model on imagenet, chop off its last classification layers (Dense Layers) and finetune it."},{"metadata":{"trusted":true},"cell_type":"code","source":"pretrained= tf.keras.applications.InceptionResNetV2(\n                include_top=False, weights='imagenet',input_shape=(*IMAGE_DIM,3)\n            )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pretrained.trainable=True","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model = tf.keras.Sequential([\n    pretrained,\n    tf.keras.layers.GlobalAveragePooling2D(),\n    tf.keras.layers.Dense(1,activation=\"sigmoid\")\n])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.compile(loss=\"binary_crossentropy\",optimizer=\"adam\",metrics=[\"accuracy\"])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Callbacks for:\n- model_checkpointing- For checpointing model with best validation accuracy.\n- early_stop- Stop training of model if model's validation accuracy did not improved in last 10 steps\n- reduce_lr- reduce learning rate if validation accuracy did not improved in last 5 steps."},{"metadata":{"trusted":true},"cell_type":"code","source":"LR_START = 0.00001\nLR_MAX = 0.00005\nLR_MIN = 0.00001\nLR_RAMPUP_EPOCHS = 5\nLR_SUSTAIN_EPOCHS = 0\nLR_EXP_DECAY = .8\n\ndef change_lr(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        lr = (LR_MAX - LR_MIN) * LR_EXP_DECAY**(epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS) + LR_MIN\n    return lr\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"checkpoint_path=\"best_checkpoint\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model_checkpoint=tf.keras.callbacks.ModelCheckpoint(checkpoint_path,monitor=\"val_accuracy\",\n                                                    save_best_only=True,mode=\"max\",\n                                                    save_weights_only=True,\n                                                    verbose=1)\nearly_stop=tf.keras.callbacks.EarlyStopping(monitor=\"val_accuracy\",patience=10,\n                                            mode=\"max\",verbose=1)\nchange_lr = tf.keras.callbacks.LearningRateScheduler(change_lr, verbose=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"callbacks=[model_checkpoint,early_stop,change_lr]","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"model training happens here with fit method. It takes tf datasets and steps."},{"metadata":{"trusted":true},"cell_type":"code","source":"history=model.fit(train_dataset,\n                  epochs=EPOCHS,\n                  steps_per_epoch=num_train_images//BATCH_SIZE,\n                  validation_data=eval_dataset,\n                  validation_steps=num_eval_images//BATCH_SIZE,\n                  callbacks=callbacks)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig, ax = plt.subplots(2,1)\nax[0].plot(history.history['loss'], color='b', label=\"Training loss\")\nax[0].plot(history.history['val_loss'], color='r', label=\"validation loss\",axes =ax[0])\nlegend = ax[0].legend(loc='best', shadow=True)\n\nax[1].plot(history.history['accuracy'], color='b', label=\"Training accuracy\")\nax[1].plot(history.history['val_accuracy'], color='r',label=\"Validation accuracy\")\nlegend = ax[1].legend(loc='best', shadow=True)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Loading best model checkpoint"},{"metadata":{"trusted":true},"cell_type":"code","source":"if os.path.isfile(checkpoint_path):\n    model.load_weights(checkpoint_path)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Predicting test data\n\nTest dataset is created which return *(batch_size,Image Tensor)* as single element."},{"metadata":{"trusted":true},"cell_type":"code","source":"test_images_path = [get_path_of_image(str(x)) for x in testX]\ntest_dataset = tf.data.Dataset.from_tensor_slices(test_images_path).map(\n        lambda path: load_tf_image(path,IMAGE_DIM),\n        num_parallel_calls=tf.data.experimental.AUTOTUNE\n)\ntest_dataset=test_dataset.batch(BATCH_SIZE)\ntest_dataset=test_dataset.prefetch(tf.data.experimental.AUTOTUNE)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"predicting test data using test dataset and saving results in`predictions`"},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions = model.predict(test_dataset,verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions= np.squeeze(predictions,axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predictions.shape","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Saving predictions is submission.csv file."},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.stable = predictions\nsubmission.to_csv('submission.csv', index=False)\nsubmission.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}