{"cells":[{"metadata":{"trusted":true,"_uuid":"91cda550d445dfc4f0948ed1678d2269e80b0cb1"},"cell_type":"code","source":"%reload_ext autoreload\n%autoreload 2\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\n\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.models import Model\nfrom keras.layers import Dense, Conv2D, MaxPooling2D, Dropout, Input, Flatten\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint, History\nfrom keras.applications.xception import Xception\nfrom keras.applications.vgg16 import VGG16\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"# First let's create a dataframes of the images and their labels"},{"metadata":{"trusted":true,"_uuid":"09980527fbf16c9037be6484aeeee9b2491b0451"},"cell_type":"code","source":"data_dir = \"../input/dogs-vs-cats-redux-kernels-edition/train/train\"\nfilenames = os.listdir(data_dir)\ndog_files = [f for f in filenames if 'dog' in f]\ncat_files = [f for f in filenames if 'cat' in f]\n\ndf = pd.DataFrame({\n    'filename': dog_files + cat_files,\n    'label': ['dog'] * len(dog_files) + ['cat'] * len(cat_files)\n})\ndf = df.sample(frac=1).reset_index(drop=True)\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"819f65d72ea9bdee1388595583de51a38cc550dd"},"cell_type":"markdown","source":"# Let's check out some of the images"},{"metadata":{"trusted":true,"_uuid":"55782aeece430565cb74c4f0248712a57820a4d2"},"cell_type":"code","source":"for i in range(5):\n    img = mpimg.imread(data_dir + '/' + df.iloc[i]['filename'])\n    plt.imshow(img)\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fb9f5866211cd9e8bdb6d7964f61bd5f357976a1"},"cell_type":"markdown","source":"# Now let's create our model"},{"metadata":{"trusted":true,"_uuid":"3f3eb2cb5e56852c32159f15c69e883d840bbd3d"},"cell_type":"code","source":"# Constants\ninput_shape = (128,128,3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0cc9cb5ba909f9554e3d9f6ffa5683db570fef64"},"cell_type":"code","source":"def get_model():\n    X_input = Input(shape=input_shape)\n    X = Conv2D(128, \n               kernel_size=(3,3), \n               padding='same', \n               activation='relu')(X_input)\n    X = MaxPooling2D(pool_size=(3,3))(X)\n    X = Dropout(0.25)(X)\n\n    X = Conv2D(256, \n               kernel_size=(3,3), \n               padding='same',\n               activation='relu')(X)\n    X = MaxPooling2D(pool_size=(3,3))(X)\n    X = Dropout(0.25)(X)\n    \n    X = Conv2D(256, \n               kernel_size=(3,3), \n               padding='same',\n               activation='relu')(X)\n    X = MaxPooling2D(pool_size=(3,3))(X)\n    X = Dropout(0.25)(X)\n    \n    X = Flatten()(X)\n    X = Dense(128, activation='relu')(X)\n#     X = Dropout(0.5)(X)\n    out = Dense(1, activation='sigmoid')(X)\n    return Model(X_input, [out])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d99c60ff89f108356f196329aae27bf7145eda42"},"cell_type":"code","source":"model = get_model()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"df04482e75765de8ecb79b42669a9e13c8f35176"},"cell_type":"markdown","source":"# Create a DataGen to pass into our model"},{"metadata":{"trusted":true,"_uuid":"9d0a50742289416e03f367dbec81e39a5a437356"},"cell_type":"code","source":"train_size = int(df.shape[0] * 0.80)\ntrain_df = df.iloc[:train_size].reset_index(drop=True)\nvalid_df = df.iloc[train_size:].reset_index(drop=True)\n\ntrain_datagen = ImageDataGenerator(\n    rescale=1./255,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    rotation_range=30,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode='nearest'\n)\nvalid_datagen = ImageDataGenerator(rescale=1./255)\n\ntrain_generator = train_datagen.flow_from_dataframe(\n    train_df,\n    data_dir,\n    x_col='filename',\n    y_col='label',\n    target_size=(input_shape[0], input_shape[1]),\n    batch_size=32,\n    class_mode='binary',\n    shuffle=True,\n    seed=42,\n)\nvalid_generator = valid_datagen.flow_from_dataframe(\n    valid_df,\n    data_dir,\n    x_col='filename',\n    y_col='label',\n    target_size=(input_shape[0], input_shape[1]),\n    batch_size=32,\n    class_mode='binary',\n    shuffle=True,\n    seed=42,\n)\n\nprint(train_df['label'].value_counts())\nprint(valid_df['label'].value_counts())\nfor c in train_generator.class_indices:\n    if train_generator.class_indices[c] != valid_generator.class_indices[c]:\n        raise ValueError(f\"Mismatching Classses: {train_generator.class_indices[c]} {valid_generator.class_indices[c]}\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"69d1c4341955a5c4c4489acffadce727d77042ff"},"cell_type":"markdown","source":"# Train the model"},{"metadata":{"trusted":true,"_uuid":"c2ebac3c3b711f0c8cbecd648491bcc9dd93d87c"},"cell_type":"code","source":"model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\nearly_stop_callback = EarlyStopping(monitor='val_loss',\n                                    verbose=1,\n                                    patience=2)\ncheckpoint_callback = ModelCheckpoint('./best-model.h5', save_best_only=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0df6ff60a38ff7e6813f30c18396ecef2817b199"},"cell_type":"code","source":"history = model.fit_generator(\n    train_generator,\n    epochs=100,\n    callbacks=[\n        early_stop_callback,\n        checkpoint_callback\n    ],\n    validation_data=valid_generator,\n    verbose=1,\n    shuffle=True,\n)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"360ddc62560ec29855c5f028c5c619fcaead0c7d"},"cell_type":"markdown","source":"# Evaluate Score"},{"metadata":{"trusted":true,"_uuid":"d17f633bd11cd65b9ffa2841a82e57c3f29e02a7"},"cell_type":"code","source":"def plot_results(h):\n    plt.plot(h['loss'], label='Train Loss')\n    plt.plot(h['val_loss'], label='Val. Loss')\n    plt.ylabel('Loss')\n    plt.xlabel('Epoch')\n    plt.legend()\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"93738ad98ee4baae55ce077bf543546a5462f321"},"cell_type":"code","source":"plot_results(history)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1b0e11a02890b7c83c53a5b248b716394469c566"},"cell_type":"markdown","source":"# Now let's try again using TransferLearning"},{"metadata":{"trusted":true,"_uuid":"0f88eecbadad6d93cd4547ae2c10deff12b07bfc"},"cell_type":"code","source":"def get_transfer_model():\n    pretrained_model = Xception(weights='../input/xception/xception_weights_tf_dim_ordering_tf_kernels_notop.h5',\n                                include_top=False,\n                                input_shape=input_shape)\n    for l in pretrained_model.layers:\n        l.trainable = False\n    \n    X_input = Input(shape=input_shape)\n    X = pretrained_model(X_input)\n    X = Flatten()(X)\n    X = Dense(256, activation='relu')(X)\n    X = Dropout(0.25)(X)\n    out = Dense(1, activation='sigmoid')(X)\n    return Model(X_input, [out])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3a7970d9026c8ce09704ff4f6a08fd91bccc7826"},"cell_type":"code","source":"transfer_model = get_transfer_model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e2fa88e1291d252f21acd413640e347ceec8fd3d"},"cell_type":"code","source":"transfer_model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\nearly_stop_callback = EarlyStopping(monitor='val_loss',\n                                    verbose=1,\n                                    patience=5,\n                                    min_delta=1e-3)\ncheckpoint_callback = ModelCheckpoint('./transfer-best-model.h5', save_best_only=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e7e0055db77e21eaf12d5ee51898a6b3446dc11e","scrolled":true},"cell_type":"code","source":"transfer_history = transfer_model.fit_generator(\n    train_generator,\n    epochs=100,\n    callbacks=[\n        early_stop_callback,\n        checkpoint_callback\n    ],\n    validation_data=valid_generator,\n    verbose=1,\n    shuffle=True,\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a0d16f642707a708a8b89ea2eb24790237d4b805"},"cell_type":"code","source":"all_history = {\n    'loss': [],\n    'val_loss': []\n}\nall_history['loss'] = all_history['loss'] + transfer_history.history['loss']\nall_history['val_loss'] = all_history['val_loss'] + transfer_history.history['val_loss']\n\nplot_results(all_history)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"91ce65b38aef24d7ec25be4d8ae9493b54e744ad"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}