{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Importing Libraries \nimport numpy as np \nimport pandas as pd \nimport tensorflow as tf\nimport matplotlib.pyplot as plt \nimport seaborn as sns\nimport tensorflow\nfrom keras.models import Sequential\nfrom tensorflow import keras\nfrom keras import layers\n\nfrom tensorflow.keras.optimizers import RMSprop\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D\nfrom tensorflow.keras.layers import MaxPool2D\nfrom tensorflow.keras.layers import Flatten\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom sklearn.model_selection import train_test_split \nimport zipfile\nfrom keras.preprocessing.image import load_img,ImageDataGenerator\n\nimport cv2\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \n        \n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-28T16:03:40.826643Z","iopub.execute_input":"2022-07-28T16:03:40.827034Z","iopub.status.idle":"2022-07-28T16:03:46.954659Z","shell.execute_reply.started":"2022-07-28T16:03:40.827005Z","shell.execute_reply":"2022-07-28T16:03:46.953605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Unzipping training file\ntrain_zip='../input/dogs-vs-cats-redux-kernels-edition/train.zip'\nzip_ref=zipfile.ZipFile(train_zip,'r')\nzip_ref.extractall('/kaggle/working/')\nzip_ref.close()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:03:46.956660Z","iopub.execute_input":"2022-07-28T16:03:46.957846Z","iopub.status.idle":"2022-07-28T16:03:58.870029Z","shell.execute_reply.started":"2022-07-28T16:03:46.957803Z","shell.execute_reply":"2022-07-28T16:03:58.869063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_dir = '/kaggle/working/'\ntrain_dir = os.path.join(base_dir, 'train')\ntrain_img_path = os.listdir(train_dir)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:03:58.871408Z","iopub.execute_input":"2022-07-28T16:03:58.871771Z","iopub.status.idle":"2022-07-28T16:03:58.892320Z","shell.execute_reply.started":"2022-07-28T16:03:58.871736Z","shell.execute_reply":"2022-07-28T16:03:58.891346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_img_path[:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:03:58.895036Z","iopub.execute_input":"2022-07-28T16:03:58.895451Z","iopub.status.idle":"2022-07-28T16:03:58.904202Z","shell.execute_reply.started":"2022-07-28T16:03:58.895415Z","shell.execute_reply":"2022-07-28T16:03:58.903255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('total training images:',len(train_img_path))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:03:58.905699Z","iopub.execute_input":"2022-07-28T16:03:58.906336Z","iopub.status.idle":"2022-07-28T16:03:58.915439Z","shell.execute_reply.started":"2022-07-28T16:03:58.906302Z","shell.execute_reply":"2022-07-28T16:03:58.914234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##Check if the training data is balanced \ncategories=[]\nfor img in train_img_path :\n    category=img.split('.')[0]\n    if category=='dog':\n        categories.append('dog')\n    else :\n        categories.append('cat')\n        \ndf=pd.DataFrame({'Image':train_img_path,'Category':categories})","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:03:58.916916Z","iopub.execute_input":"2022-07-28T16:03:58.917221Z","iopub.status.idle":"2022-07-28T16:03:58.943305Z","shell.execute_reply.started":"2022-07-28T16:03:58.917198Z","shell.execute_reply":"2022-07-28T16:03:58.942213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,10))\nsns.countplot(data=df,x='Category');","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:03:58.944915Z","iopub.execute_input":"2022-07-28T16:03:58.945373Z","iopub.status.idle":"2022-07-28T16:03:59.164813Z","shell.execute_reply.started":"2022-07-28T16:03:58.945337Z","shell.execute_reply":"2022-07-28T16:03:59.163837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##plot random image from the training set\nimport random\nsample=random.choice(train_img_path)\nplt.imshow(plt.imread('/kaggle/working/train/'+sample));","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:03:59.166886Z","iopub.execute_input":"2022-07-28T16:03:59.167555Z","iopub.status.idle":"2022-07-28T16:03:59.398860Z","shell.execute_reply.started":"2022-07-28T16:03:59.167518Z","shell.execute_reply":"2022-07-28T16:03:59.397853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Splitting the training data to train and validation \nfrom sklearn.model_selection import train_test_split \ntrain,validation=train_test_split(df,test_size=0.2,random_state=43)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:03:59.399993Z","iopub.execute_input":"2022-07-28T16:03:59.400391Z","iopub.status.idle":"2022-07-28T16:03:59.411653Z","shell.execute_reply.started":"2022-07-28T16:03:59.400350Z","shell.execute_reply":"2022-07-28T16:03:59.410791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:03:59.416427Z","iopub.execute_input":"2022-07-28T16:03:59.416710Z","iopub.status.idle":"2022-07-28T16:03:59.430805Z","shell.execute_reply.started":"2022-07-28T16:03:59.416682Z","shell.execute_reply":"2022-07-28T16:03:59.429793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Rename the index of train and validation \ntrain=train.reset_index(drop=True)\nvalidation=validation.reset_index(drop=True)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:03:59.432187Z","iopub.execute_input":"2022-07-28T16:03:59.433058Z","iopub.status.idle":"2022-07-28T16:03:59.443906Z","shell.execute_reply.started":"2022-07-28T16:03:59.433025Z","shell.execute_reply":"2022-07-28T16:03:59.442859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Training Generator**","metadata":{}},{"cell_type":"code","source":"## Applying Image augmentation on training set \nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\ntrain_gen=ImageDataGenerator(rotation_range=15,\n    rescale=1./255,\n    shear_range=0.1,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    width_shift_range=0.1,\n    height_shift_range=0.1)\n\n\ntrain_generator = train_gen.flow_from_dataframe(train,\n                                                    directory=\"./train\",\n                                                    x_col='Image',\n                                                    y_col='Category',\n                                                    batch_size=20,\n                                                    class_mode='binary',\n                                                    target_size=(150, 150)) \n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:03:59.445411Z","iopub.execute_input":"2022-07-28T16:03:59.445881Z","iopub.status.idle":"2022-07-28T16:03:59.643849Z","shell.execute_reply.started":"2022-07-28T16:03:59.445845Z","shell.execute_reply":"2022-07-28T16:03:59.642940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Validation Generator**","metadata":{}},{"cell_type":"code","source":"## Applying Image Augmentation on the validation set \nvalid_gen  = ImageDataGenerator( rescale = 1.0/255.)\nvalidation_generator =  valid_gen.flow_from_dataframe(validation,\n                                                            directory=\"./train\",\n                                                              x_col='Image',\n                                                             y_col='Category',\n                                                              batch_size=20,\n                                                              class_mode  = 'binary',\n                                                              target_size = (150, 150))","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:03:59.645186Z","iopub.execute_input":"2022-07-28T16:03:59.645864Z","iopub.status.idle":"2022-07-28T16:03:59.703982Z","shell.execute_reply.started":"2022-07-28T16:03:59.645828Z","shell.execute_reply":"2022-07-28T16:03:59.702986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Visualize some example from generator**\n","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12, 12))\nfor i in range(0, 15):\n    plt.subplot(5, 3, i+1)\n    for X_batch, Y_batch in train_generator:\n        image = X_batch[0]\n        plt.imshow(image)\n        break\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:03:59.705524Z","iopub.execute_input":"2022-07-28T16:03:59.705870Z","iopub.status.idle":"2022-07-28T16:04:03.221871Z","shell.execute_reply.started":"2022-07-28T16:03:59.705837Z","shell.execute_reply":"2022-07-28T16:04:03.221015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Load Inception pretrained model and rebuild model**","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Input, Dense, Dropout, Flatten, MaxPool2D\nfrom keras.applications.xception import preprocess_input\n\nxception_model = tf.keras.applications.xception.Xception(\n    include_top=False,\n    weights='imagenet',\n    input_tensor=None,\n    input_shape=(150,150, 3),\n    pooling='avg',\n    classes=1,\n    classifier_activation='sigmoid'\n    )\ntf.random.set_seed(24)\nmodel = Sequential()\nmodel.add(xception_model)\nmodel.add(Flatten()) \nmodel.add(Dense(units=2048, activation=\"relu\"))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(units=1024, activation='relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(units=512, activation='relu'))\nmodel.add(Dropout(0.2))\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.layers[0].trainable=False\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:04:03.223456Z","iopub.execute_input":"2022-07-28T16:04:03.224111Z","iopub.status.idle":"2022-07-28T16:04:10.553972Z","shell.execute_reply.started":"2022-07-28T16:04:03.224065Z","shell.execute_reply":"2022-07-28T16:04:10.552998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='binary_crossentropy', optimizer=tf.keras.optimizers.Adam(learning_rate=0.0001), metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:04:10.555570Z","iopub.execute_input":"2022-07-28T16:04:10.555912Z","iopub.status.idle":"2022-07-28T16:04:10.572151Z","shell.execute_reply.started":"2022-07-28T16:04:10.555877Z","shell.execute_reply":"2022-07-28T16:04:10.571133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint\nearly_stopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=5)\nsave_best = ModelCheckpoint(\nfilepath = 'best_model.hdf5',\nverbose=1, save_best_only=True\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:04:10.573507Z","iopub.execute_input":"2022-07-28T16:04:10.574443Z","iopub.status.idle":"2022-07-28T16:04:10.580381Z","shell.execute_reply.started":"2022-07-28T16:04:10.574408Z","shell.execute_reply":"2022-07-28T16:04:10.579403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size=32\nepochs=15\n\nhistory = model.fit(train_generator, steps_per_epoch=train_generator.samples // batch_size, validation_data = validation_generator,\\\n                         validation_steps = validation_generator.samples // batch_size, epochs = epochs, callbacks=[save_best,early_stopping], verbose=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:04:10.581791Z","iopub.execute_input":"2022-07-28T16:04:10.582696Z","iopub.status.idle":"2022-07-28T16:18:36.729274Z","shell.execute_reply.started":"2022-07-28T16:04:10.582657Z","shell.execute_reply":"2022-07-28T16:18:36.727904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:18:36.730381Z","iopub.status.idle":"2022-07-28T16:18:36.731498Z","shell.execute_reply.started":"2022-07-28T16:18:36.731227Z","shell.execute_reply":"2022-07-28T16:18:36.731255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:18:36.732776Z","iopub.status.idle":"2022-07-28T16:18:36.733798Z","shell.execute_reply.started":"2022-07-28T16:18:36.733518Z","shell.execute_reply":"2022-07-28T16:18:36.733545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_model = tf.keras.models.load_model('best_model.hdf5')\nbest_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:18:36.735275Z","iopub.status.idle":"2022-07-28T16:18:36.736124Z","shell.execute_reply.started":"2022-07-28T16:18:36.735847Z","shell.execute_reply":"2022-07-28T16:18:36.735871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights('model_catsvsdogs.h5')\nmodel.save('model_catsvsdogs.h5')","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:18:36.737429Z","iopub.status.idle":"2022-07-28T16:18:36.738360Z","shell.execute_reply.started":"2022-07-28T16:18:36.737990Z","shell.execute_reply":"2022-07-28T16:18:36.738014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_zip='../input/dogs-vs-cats-redux-kernels-edition/test.zip'\nzip_refe=zipfile.ZipFile(test_zip,'r')\nzip_refe.extractall('/kaggle/working/')\nzip_refe.close()","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:18:36.739673Z","iopub.status.idle":"2022-07-28T16:18:36.740494Z","shell.execute_reply.started":"2022-07-28T16:18:36.740233Z","shell.execute_reply":"2022-07-28T16:18:36.740257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_image_files = os.listdir('/kaggle/working/test')\ntest_df = pd.DataFrame(data = test_image_files, columns = ['filename'])\ntest_df['id'] = test_df['filename'].apply(lambda f: int(f.split('.')[0]))\ntest_df.sort_values(by = 'id', inplace = True, ignore_index = True)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:18:36.741804Z","iopub.status.idle":"2022-07-28T16:18:36.742734Z","shell.execute_reply.started":"2022-07-28T16:18:36.742490Z","shell.execute_reply":"2022-07-28T16:18:36.742512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_gen = ImageDataGenerator(rescale = 1./255)\ntest_generator = test_gen.flow_from_dataframe(\n    test_df, \n    '/kaggle/working/test', \n    x_col='filename',\n    class_mode= None,\n    target_size=(150,150),\n    batch_size=batch_size,\n    shuffle=False\n)\npredict = best_model.predict(test_generator,batch_size=32,max_queue_size=1, verbose = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:18:36.744088Z","iopub.status.idle":"2022-07-28T16:18:36.744890Z","shell.execute_reply.started":"2022-07-28T16:18:36.744635Z","shell.execute_reply":"2022-07-28T16:18:36.744659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df[\"label\"] = predict\nresult = test_df[[\"id\", \"label\"]]\nresult.to_csv('submission_1.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-28T16:18:36.746322Z","iopub.status.idle":"2022-07-28T16:18:36.747256Z","shell.execute_reply.started":"2022-07-28T16:18:36.747008Z","shell.execute_reply":"2022-07-28T16:18:36.747031Z"},"trusted":true},"execution_count":null,"outputs":[]}]}