{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import zipfile\nimport numpy as np \nimport pandas as pd \nimport tensorflow as tf\nfrom tensorflow.keras.optimizers import RMSprop\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport cv2\nfrom keras.preprocessing import image\nfrom random import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.applications.inception_v3 import InceptionV3\nfrom tensorflow.keras import layers","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH = '/kaggle/input/dogs-vs-cats-redux-kernels-edition/'\nTRAIN_DIR = './train/'\nTEST_DIR = './test/'\nimg_width = 150\nimg_height = 150","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image_path = os.path.join(PATH, \"train.zip\")\ntest_image_path = os.path.join(PATH, \"test.zip\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Unzip the train and test files.**","metadata":{}},{"cell_type":"code","source":"with zipfile.ZipFile(train_image_path, \"r\") as z:\n    z.extractall(\".\")\n    \nwith zipfile.ZipFile(test_image_path, \"r\") as z:\n    z.extractall(\".\") ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_images_list = [TRAIN_DIR + i for i in os.listdir(TRAIN_DIR)] \ntest_images_list = [TEST_DIR + i for i in os.listdir(TEST_DIR)]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We select 7500 cats and 7500 dogs images.**","metadata":{}},{"cell_type":"code","source":"cats_imgs = []\ndogs_imgs = []\n\n\ndef img_seperator(img_list):\n    for img in train_images_list:\n        if 'cat' in img:\n            cats_imgs.append(img)\n        elif 'dog' in img:\n            dogs_imgs.append(img)\n    return cats_imgs, dogs_imgs\n\nimg_seperator(test_images_list)\n\n\ncats_imgs = cats_imgs[:7500]\ndogs_imgs = dogs_imgs[:7500]\n\ntrain_images_list = cats_imgs + dogs_imgs\n\nprint(len(train_images_list))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_data(list_of_images):\n    x = []  # images as arrays\n    y = []  # labels\n    \n    for img in list_of_images:\n        img = cv2.imread(img)\n        img = cv2.resize(img, (img_width,img_height))    \n        x.append(img)\n    \n    for i in list_of_images:\n        if 'dog' in i:\n            y.append(1)\n        elif 'cat' in i:\n            y.append(0)\n            \n    return x, y      ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X, y = process_data(train_images_list)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**CNN InceptionV3 model!**","metadata":{}},{"cell_type":"code","source":"local_weights_file = \"../input/inceptionv3/inception_v3_weights_tf_dim_ordering_tf_kernels_notop.h5\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize the base model.\n# Set the input shape and remove the dense layers.\npre_trained_model = InceptionV3(input_shape = (150, 150, 3), \n                                include_top = False, \n                                weights = None)\n\n# Load the pre-trained weights you downloaded.\npre_trained_model.load_weights(local_weights_file)\n\n# Freeze the weights of the layers.\nfor layer in pre_trained_model.layers:\n    layer.trainable = False\n\n\n#pre_trained_model.summary()\n\n# Choose `mixed_7` as the last layer of your base model\nlast_layer = pre_trained_model.get_layer('mixed7')\nprint('last layer output shape: ', last_layer.output_shape)\nlast_output = last_layer.output","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Add dense layers to the classifier**\n\n**Next, we will add dense layers to your model. These will be the layers that we will train and is tasked with recognizing cats and dogs. We will add a Dropout layer as well to regularize the output and avoid overfitting.**","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import Model\n\n# Flatten the output layer to 1 dimension\nx = layers.Flatten()(last_output)\n# Add a fully connected layer with 1,024 hidden units and ReLU activation\nx = layers.Dense(1024, activation='relu')(x)\n# Add a dropout rate of 0.2\nx = layers.Dropout(0.5)(x)                  \n# Add a final sigmoid layer for classification\nx = layers.Dense(1, activation='sigmoid')(x)           \n\n# Append the dense network to the base model\nmodel = Model(pre_trained_model.input, x) \n\n# Print the model summary. See your dense network connected at the end.\nmodel.summary()\n\n# Set the training parameters\nmodel.compile(optimizer = RMSprop(learning_rate=0.0001), \n              loss = 'binary_crossentropy', \n              metrics = ['accuracy'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class myCallback(tf.keras.callbacks.Callback):\n    def on_epoch_end(self, epoch, logs={}):\n        if(logs.get('accuracy')>0.95):\n            print(\"\\nReached 99.9% accuracy so cancelling training!\")\n            self.model.stop_training = True","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# All images will be rescaled by 1./255 and we also rotate and do other operations\n\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    rotation_range=40,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest')\n\n\nval_datagen = ImageDataGenerator(rescale=1./255)\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Flow training images in batches of 20 using train_datagen generator\n\ntrain_generator = train_datagen.flow(\n        np.array(X_train),\n        y_train,\n        batch_size=20)\n\nvalidation_generator = val_datagen.flow(\n        np.array(X_val),\n        y_val,\n        batch_size=20)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train your model\n\ncallbacks = myCallback()\nhistory = model.fit(train_generator,\n                    steps_per_epoch=100,\n                    epochs=10,\n                    validation_data=validation_generator,\n                    validation_steps=50,\n                    callbacks=callbacks\n                   )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights('model_weights.h5')\nmodel.save('model_keras.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**After the training and the validating steps, we are ready to make the test set.**","metadata":{}},{"cell_type":"code","source":"X_test, y_test = process_data(test_images_list) #Y_test in this case will be []\n\ntest_datagen = ImageDataGenerator(rescale=1. / 255)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_generator = val_datagen.flow(np.array(X_test), batch_size=20)\n\nprediction_probabilities = model.predict_generator(test_generator, verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Ready for Submission**","metadata":{}},{"cell_type":"code","source":"counter = range(1, len(test_images_list) + 1)\nsolution = pd.DataFrame({\"id\": counter, \"label\":list(prediction_probabilities)})\ncols = ['label']\n\nfor col in cols:\n    solution[col] = solution[col].map(lambda x: str(x).lstrip('[').rstrip(']')).astype(float)\n\nsolution.to_csv(\"submission.csv\", index = False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}