{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-06T08:45:32.746748Z","iopub.execute_input":"2022-08-06T08:45:32.747340Z","iopub.status.idle":"2022-08-06T08:45:32.764730Z","shell.execute_reply.started":"2022-08-06T08:45:32.747304Z","shell.execute_reply":"2022-08-06T08:45:32.763616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport os\nimport random\nimport zipfile\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import ConfusionMatrixDisplay, confusion_matrix, classification_report \nimport tensorflow as tf\nfrom keras.preprocessing.image import load_img, ImageDataGenerator\nfrom keras.models import Sequential, Model\nfrom keras.layers import Conv2D, MaxPool2D, Flatten, Dense, GlobalAveragePooling2D, Dropout, BatchNormalization, Input\nfrom keras.callbacks import ReduceLROnPlateau, EarlyStopping, ModelCheckpoint\nfrom keras.applications.vgg16 import VGG16, preprocess_input\n\nbatch_size = 128","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:32.767003Z","iopub.execute_input":"2022-08-06T08:45:32.767491Z","iopub.status.idle":"2022-08-06T08:45:32.776535Z","shell.execute_reply.started":"2022-08-06T08:45:32.767452Z","shell.execute_reply":"2022-08-06T08:45:32.775319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed = 666\ntf.random.set_seed(seed)\nnp.random.seed(seed)\nos.environ[\"PYTHONHASHSEED\"] = str(seed)                      \nrandom.seed(666)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:32.778817Z","iopub.execute_input":"2022-08-06T08:45:32.780008Z","iopub.status.idle":"2022-08-06T08:45:32.790624Z","shell.execute_reply.started":"2022-08-06T08:45:32.779964Z","shell.execute_reply":"2022-08-06T08:45:32.789293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir(\"../input/dogs-vs-cats/\")\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:32.793384Z","iopub.execute_input":"2022-08-06T08:45:32.794089Z","iopub.status.idle":"2022-08-06T08:45:32.805913Z","shell.execute_reply.started":"2022-08-06T08:45:32.793916Z","shell.execute_reply":"2022-08-06T08:45:32.804013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_PATH = \"../input/dogs-vs-cats/train.zip\"\nTEST_PATH = \"../input/dogs-vs-cats/test1.zip\"\n\nFILES = \"/kaggle/files/unzipped/\"\n\nwith zipfile.ZipFile(TRAIN_PATH, 'r') as zipp:\n    zipp.extractall(FILES)\n    \nwith zipfile.ZipFile(TEST_PATH, 'r') as zipp:\n    zipp.extractall(FILES)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:32.807405Z","iopub.execute_input":"2022-08-06T08:45:32.808667Z","iopub.status.idle":"2022-08-06T08:45:49.718099Z","shell.execute_reply.started":"2022-08-06T08:45:32.808623Z","shell.execute_reply":"2022-08-06T08:45:49.717124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.DataFrame({\"file\": os.listdir(\"/kaggle/files/unzipped/train\")})\ntrain_df[\"label\"] = train_df[\"file\"].apply(lambda x: x.split(\".\")[0])\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:49.719272Z","iopub.execute_input":"2022-08-06T08:45:49.719792Z","iopub.status.idle":"2022-08-06T08:45:49.780834Z","shell.execute_reply.started":"2022-08-06T08:45:49.719756Z","shell.execute_reply":"2022-08-06T08:45:49.779826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.DataFrame({\"file\": os.listdir(\"/kaggle/files/unzipped/test1\")})\n\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:49.785448Z","iopub.execute_input":"2022-08-06T08:45:49.786244Z","iopub.status.idle":"2022-08-06T08:45:49.871861Z","shell.execute_reply.started":"2022-08-06T08:45:49.786206Z","shell.execute_reply":"2022-08-06T08:45:49.870868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize = (6, 6), facecolor = \"#e5e5e5\")\nax.set_facecolor(\"#e5e5e5\")\n\nsns.countplot(x = \"label\", data = train_df, ax = ax)\n\nax.set_title(\"Distribution of Class Labels\")\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:49.873172Z","iopub.execute_input":"2022-08-06T08:45:49.874149Z","iopub.status.idle":"2022-08-06T08:45:50.052124Z","shell.execute_reply.started":"2022-08-06T08:45:49.874108Z","shell.execute_reply":"2022-08-06T08:45:50.051182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(1, figsize = (8, 8))\nfig.suptitle(\"Training Set Images (Sample)\")\n\nfor i in range(25):\n\n    plt.subplot(5, 5, i + 1)\n    image = load_img(FILES + \"train/\" + train_df[\"file\"][i])\n    plt.imshow(image)\n    plt.axis(\"off\")\n    \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:50.055948Z","iopub.execute_input":"2022-08-06T08:45:50.056951Z","iopub.status.idle":"2022-08-06T08:45:51.628036Z","shell.execute_reply.started":"2022-08-06T08:45:50.056892Z","shell.execute_reply":"2022-08-06T08:45:51.627198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(1, figsize = (8, 8))\nfig.suptitle(\"Sample Dog images from Training Set\")\n\nfor i in range(25):\n    \n    plt.subplot(5, 5, i + 1)\n    image = load_img(FILES + \"train/\" + train_df.query(\"label == 'dog'\").file.values[i])\n    plt.imshow(image)\n    plt.axis(\"off\")\n    \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:51.629352Z","iopub.execute_input":"2022-08-06T08:45:51.630011Z","iopub.status.idle":"2022-08-06T08:45:53.135454Z","shell.execute_reply.started":"2022-08-06T08:45:51.629975Z","shell.execute_reply":"2022-08-06T08:45:53.132962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(1, figsize = (8, 8))\nfig.suptitle(\"Sample Cat images from Training Set\")\n\nfor i in range(25):\n    \n    plt.subplot(5, 5, i + 1)\n    image = load_img(FILES + \"train/\" + train_df.query(\"label == 'cat'\").file.values[i])\n    plt.imshow(image)\n    plt.axis(\"off\")\n    \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:53.137000Z","iopub.execute_input":"2022-08-06T08:45:53.137972Z","iopub.status.idle":"2022-08-06T08:45:55.075511Z","shell.execute_reply.started":"2022-08-06T08:45:53.137935Z","shell.execute_reply":"2022-08-06T08:45:55.073550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data, val_data = train_test_split(train_df, \n                                        test_size = 0.2, \n                                        stratify = train_df[\"label\"], \n                                        random_state = 666)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:55.076918Z","iopub.execute_input":"2022-08-06T08:45:55.078074Z","iopub.status.idle":"2022-08-06T08:45:55.119453Z","shell.execute_reply.started":"2022-08-06T08:45:55.078021Z","shell.execute_reply":"2022-08-06T08:45:55.118609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datagen = ImageDataGenerator(\n    rotation_range = 30, \n    width_shift_range = 0.1,\n    height_shift_range = 0.1, \n    brightness_range = (0.5, 1), \n    zoom_range = 0.2,\n    horizontal_flip = True, \n    rescale = 1./255,\n)\n\nsample_df = train_data.sample(1)\n\nsample_generator = datagen.flow_from_dataframe(\n    dataframe = sample_df,\n    directory = FILES + \"train/\",\n    x_col = \"file\",\n    y_col = \"label\",\n    class_mode = \"categorical\",\n    target_size = (224, 224),\n    seed = 666\n)\n\nplt.figure(figsize = (14, 8))\n\nfor i in range(50):\n    \n    plt.subplot(5, 10, i + 1)\n    \n    for X, y in sample_generator:\n\n        plt.imshow(X[0])\n        plt.axis(\"off\")\n        break\n        \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:55.120793Z","iopub.execute_input":"2022-08-06T08:45:55.121238Z","iopub.status.idle":"2022-08-06T08:45:58.215831Z","shell.execute_reply.started":"2022-08-06T08:45:55.121202Z","shell.execute_reply":"2022-08-06T08:45:58.215027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rotation_range = 15, \n#     width_shift_range = 0.1,\n#     height_shift_range = 0.1, \n#     brightness_range = (0.5, 1), \n#     zoom_range = 0.1,\n    horizontal_flip = True,\n    preprocessing_function = preprocess_input\n)\n\nval_datagen = ImageDataGenerator(preprocessing_function = preprocess_input)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:58.216992Z","iopub.execute_input":"2022-08-06T08:45:58.218119Z","iopub.status.idle":"2022-08-06T08:45:58.223653Z","shell.execute_reply.started":"2022-08-06T08:45:58.218081Z","shell.execute_reply":"2022-08-06T08:45:58.222772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = train_datagen.flow_from_dataframe(\n    dataframe = train_data,\n    directory = FILES + \"train/\",\n    x_col = \"file\",\n    y_col = \"label\",\n    class_mode = \"categorical\",\n    target_size = (224, 224),\n    batch_size = batch_size,\n    seed = 666,\n)\n\nval_generator = val_datagen.flow_from_dataframe(\n    dataframe = val_data,\n    directory = FILES + \"train/\",\n    x_col = \"file\",\n    y_col = \"label\",\n    class_mode = \"categorical\",\n    target_size = (224, 224),\n    batch_size = batch_size,\n    seed = 666,\n    shuffle = False\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:58.225577Z","iopub.execute_input":"2022-08-06T08:45:58.226387Z","iopub.status.idle":"2022-08-06T08:45:58.492997Z","shell.execute_reply.started":"2022-08-06T08:45:58.226347Z","shell.execute_reply":"2022-08-06T08:45:58.491888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.applications.vgg16 import VGG16\nfrom keras.models import Model\n\n\ninput_shape=(224,224,3)\nbatch_size= 128\n\nbase_model = VGG16(\n    weights = \"imagenet\", \n    input_shape = (224, 224, 3),\n    include_top = False\n)\n\nfor layer in base_model.layers:\n    layer.trainable = False\n    \ndef vgg16_pretrained():\n    \n    model= Sequential([\n        base_model,\n        GlobalAveragePooling2D(),\n        Dense(100,activation='relu'),\n        Dropout(0.4),\n        Dense(64,activation='relu'),\n        Dense(2,activation='softmax')\n    ])\n    \n    return model\n\n\ntf.keras.backend.clear_session()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:58.494330Z","iopub.execute_input":"2022-08-06T08:45:58.496154Z","iopub.status.idle":"2022-08-06T08:45:58.775811Z","shell.execute_reply.started":"2022-08-06T08:45:58.496115Z","shell.execute_reply":"2022-08-06T08:45:58.774928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = vgg16_pretrained()\n\nmodel.compile(loss = \"categorical_crossentropy\", optimizer = \"adam\", metrics = \"accuracy\")\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:58.779899Z","iopub.execute_input":"2022-08-06T08:45:58.780183Z","iopub.status.idle":"2022-08-06T08:45:58.867278Z","shell.execute_reply.started":"2022-08-06T08:45:58.780157Z","shell.execute_reply":"2022-08-06T08:45:58.866293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reduce_lr = ReduceLROnPlateau(\n    monitor = \"val_accuracy\", \n    patience = 2,\n    verbose = 1, \n    factor = 0.5, \n    min_lr = 0.000000001\n)\n\nearly_stopping = EarlyStopping(\n    monitor = \"val_accuracy\",\n    patience = 5,\n    verbose = 1,\n    mode = \"max\",\n)\n\ncheckpoint = ModelCheckpoint(\n    monitor = \"val_accuracy\",\n    filepath = \"catdog_vgg16_.{epoch:02d}-{val_accuracy:.6f}.hdf5\",\n    verbose = 1,\n    save_best_only = True, \n    save_weights_only = True\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:58.868777Z","iopub.execute_input":"2022-08-06T08:45:58.870118Z","iopub.status.idle":"2022-08-06T08:45:58.876754Z","shell.execute_reply.started":"2022-08-06T08:45:58.870080Z","shell.execute_reply":"2022-08-06T08:45:58.875681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_generator,\n    epochs = 10, \n    validation_data = val_generator,\n    validation_steps = val_data.shape[0] // batch_size,\n    steps_per_epoch = train_data.shape[0] // batch_size,\n    callbacks = [reduce_lr, early_stopping, checkpoint]\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T08:45:58.878277Z","iopub.execute_input":"2022-08-06T08:45:58.878635Z","iopub.status.idle":"2022-08-06T09:38:37.144189Z","shell.execute_reply.started":"2022-08-06T08:45:58.878598Z","shell.execute_reply":"2022-08-06T09:38:37.143160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nmodel = vgg16_pretrained()\n\nmodel.load_weights(\"./catdog_vgg16_.10-0.983774.hdf5\")","metadata":{"execution":{"iopub.status.busy":"2022-08-06T09:38:37.145682Z","iopub.execute_input":"2022-08-06T09:38:37.146067Z","iopub.status.idle":"2022-08-06T09:38:37.287808Z","shell.execute_reply.started":"2022-08-06T09:38:37.146029Z","shell.execute_reply":"2022-08-06T09:38:37.286799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 2, figsize = (12, 4))\n\nsns.lineplot(x = range(len(history.history[\"loss\"])), y = history.history[\"loss\"], ax = axes[0], label = \"Training Loss\")\nsns.lineplot(x = range(len(history.history[\"loss\"])), y = history.history[\"val_loss\"], ax = axes[0], label = \"Validation Loss\")\n\nsns.lineplot(x = range(len(history.history[\"accuracy\"])), y = history.history[\"accuracy\"], ax = axes[1], label = \"Training Accuracy\")\nsns.lineplot(x = range(len(history.history[\"accuracy\"])), y = history.history[\"val_accuracy\"], ax = axes[1], label = \"Validation Accuracy\")\naxes[0].set_title(\"Loss\"); axes[1].set_title(\"Accuracy\")\n\nsns.despine()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T09:38:37.289202Z","iopub.execute_input":"2022-08-06T09:38:37.289556Z","iopub.status.idle":"2022-08-06T09:38:37.652712Z","shell.execute_reply.started":"2022-08-06T09:38:37.289522Z","shell.execute_reply":"2022-08-06T09:38:37.651781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_pred = model.predict(val_generator, steps = np.ceil(val_data.shape[0] / batch_size))\nval_data.loc[:, \"val_pred\"] = np.argmax(val_pred, axis = 1)\n\nlabels = dict((v, k) for k, v in val_generator.class_indices.items())\n\nval_data.loc[:, \"val_pred\"] = val_data.loc[:, \"val_pred\"].map(labels)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T09:38:37.657041Z","iopub.execute_input":"2022-08-06T09:38:37.658090Z","iopub.status.idle":"2022-08-06T09:38:59.853647Z","shell.execute_reply.started":"2022-08-06T09:38:37.658051Z","shell.execute_reply":"2022-08-06T09:38:59.852691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize = (9, 6))\n\ncm = confusion_matrix(val_data[\"label\"], val_data[\"val_pred\"])\n\ndisp = ConfusionMatrixDisplay(confusion_matrix = cm, display_labels = [\"cat\", \"dog\"])\ndisp.plot(cmap = plt.cm.Blues, ax = ax)\n\nax.set_title(\"Validation Set\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T09:38:59.855298Z","iopub.execute_input":"2022-08-06T09:38:59.856033Z","iopub.status.idle":"2022-08-06T09:39:00.093821Z","shell.execute_reply.started":"2022-08-06T09:38:59.855995Z","shell.execute_reply":"2022-08-06T09:39:00.092915Z"},"trusted":true},"execution_count":null,"outputs":[]}]}