{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-08T05:50:11.352850Z","iopub.execute_input":"2022-08-08T05:50:11.353796Z","iopub.status.idle":"2022-08-08T05:50:11.386707Z","shell.execute_reply.started":"2022-08-08T05:50:11.353666Z","shell.execute_reply":"2022-08-08T05:50:11.385525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_dir = \"/kaggle/input/dogs-vs-cats/\"\ntrain_dir = os.path.join(base_dir, \"train.zip\")\ntest_dir = os.path.join(base_dir, \"test1.zip\")\n\nimport zipfile\nwith zipfile.ZipFile(train_dir,\"r\") as z:\n    z.extractall()\n\nwith zipfile.ZipFile(test_dir,\"r\") as z:\n    z.extractall()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T05:50:14.083726Z","iopub.execute_input":"2022-08-08T05:50:14.084400Z","iopub.status.idle":"2022-08-08T05:50:34.863464Z","shell.execute_reply.started":"2022-08-08T05:50:14.084339Z","shell.execute_reply":"2022-08-08T05:50:34.861007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = os.listdir(\"./train\")\ndata = pd.DataFrame(images)\ndata = data.rename(columns={0: \"image\"})\ndata['image'] = data['image'].apply(lambda x: \"./train/\"+x)\ndata['label'] = data['image'].apply(lambda x: 0 if 'cat' in x else 1)\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T05:50:34.865711Z","iopub.execute_input":"2022-08-08T05:50:34.866200Z","iopub.status.idle":"2022-08-08T05:50:34.943429Z","shell.execute_reply.started":"2022-08-08T05:50:34.866150Z","shell.execute_reply":"2022-08-08T05:50:34.942304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, BatchNormalization\nfrom keras.layers import Conv2D, MaxPooling2D\nfrom keras.preprocessing import image\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.utils import to_categorical\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:20:39.693342Z","iopub.execute_input":"2022-08-08T06:20:39.694111Z","iopub.status.idle":"2022-08-08T06:20:39.701391Z","shell.execute_reply.started":"2022-08-08T06:20:39.694058Z","shell.execute_reply":"2022-08-08T06:20:39.700544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_image = []\nfor i in tqdm(range(data.shape[0])):\n    img = image.load_img(data['image'][i], target_size=(64,64,1), color_mode = \"grayscale\")\n    img = image.img_to_array(img)\n    img = img/255\n    train_image.append(img)\nX = np.array(train_image)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T05:50:47.257370Z","iopub.execute_input":"2022-08-08T05:50:47.258839Z","iopub.status.idle":"2022-08-08T05:52:11.298254Z","shell.execute_reply.started":"2022-08-08T05:50:47.258784Z","shell.execute_reply":"2022-08-08T05:52:11.296605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=data['label'].values\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T05:53:13.533067Z","iopub.execute_input":"2022-08-08T05:53:13.533665Z","iopub.status.idle":"2022-08-08T05:53:13.544049Z","shell.execute_reply.started":"2022-08-08T05:53:13.533619Z","shell.execute_reply":"2022-08-08T05:53:13.542470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, random_state=42, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:05:31.417182Z","iopub.execute_input":"2022-08-08T06:05:31.417673Z","iopub.status.idle":"2022-08-08T06:05:31.750305Z","shell.execute_reply.started":"2022-08-08T06:05:31.417636Z","shell.execute_reply":"2022-08-08T06:05:31.749308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(batch_norm=False):\n    model = Sequential()\n    model.add(Conv2D(64, kernel_size=(3,3), input_shape=(64, 64, 1), activation='relu', kernel_initializer='he_normal'))\n    if batch_norm:\n        model.add(BatchNormalization())\n    model.add(MaxPooling2D(pool_size=3))\n    model.add(Dropout(0.4))\n    model.add(Conv2D(32, kernel_size=(3,3), activation = 'relu'))\n    if batch_norm:\n        model.add(BatchNormalization())\n    model.add(MaxPooling2D(pool_size=3))\n    model.add(Dropout(0.5))\n    model.add(Flatten())\n    model.add(Dense(128, activation='relu'))\n    model.add(Dropout(0.4))\n    model.add(Dense(1, activation='sigmoid'))\n    print(model.summary())\n    model.compile(loss='binary_crossentropy',optimizer='Adam',metrics=['accuracy'])    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:20:27.809151Z","iopub.execute_input":"2022-08-08T06:20:27.810476Z","iopub.status.idle":"2022-08-08T06:20:27.821739Z","shell.execute_reply.started":"2022-08-08T06:20:27.810423Z","shell.execute_reply":"2022-08-08T06:20:27.820439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = train()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:05:30.005937Z","iopub.execute_input":"2022-08-08T06:05:30.007713Z","iopub.status.idle":"2022-08-08T06:05:30.107806Z","shell.execute_reply.started":"2022-08-08T06:05:30.007648Z","shell.execute_reply":"2022-08-08T06:05:30.106477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training = model.fit(X_train, y_train, epochs=25, validation_data=(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:05:43.374150Z","iopub.execute_input":"2022-08-08T06:05:43.374688Z","iopub.status.idle":"2022-08-08T06:19:57.309993Z","shell.execute_reply.started":"2022-08-08T06:05:43.374650Z","shell.execute_reply":"2022-08-08T06:19:57.308700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ndef plot_acc_loss(model_training_history, keyword =\"accuracy\", label=\"\"):\n    # summarize history for accuracy\n    plt.plot(training.history[keyword])\n    plt.plot(training.history[f'val_{keyword}'])\n    title = f'model {keyword} {label}'.strip()\n    plt.title(title)\n    plt.ylabel(keyword)\n    plt.xlabel('epoch')\n    plt.legend(['train', 'test'], loc='upper left')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:20:03.795026Z","iopub.execute_input":"2022-08-08T06:20:03.795539Z","iopub.status.idle":"2022-08-08T06:20:03.804403Z","shell.execute_reply.started":"2022-08-08T06:20:03.795498Z","shell.execute_reply":"2022-08-08T06:20:03.802925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Use of dropout ensures a more robust model at test time\n\n#### Training with batchnormalization to improve perfomance and speed","metadata":{}},{"cell_type":"code","source":"model = train(batch_norm=True)\ntraining_batchnorm = model.fit(X_train, y_train, epochs=25, validation_data=(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:20:43.857141Z","iopub.execute_input":"2022-08-08T06:20:43.858085Z","iopub.status.idle":"2022-08-08T06:40:19.530504Z","shell.execute_reply.started":"2022-08-08T06:20:43.858039Z","shell.execute_reply":"2022-08-08T06:40:19.529205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_acc_loss(training, label=\"no batch norm\")","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:40:53.857114Z","iopub.execute_input":"2022-08-08T06:40:53.857680Z","iopub.status.idle":"2022-08-08T06:40:54.104571Z","shell.execute_reply.started":"2022-08-08T06:40:53.857638Z","shell.execute_reply":"2022-08-08T06:40:54.103002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_acc_loss(training_batchnorm, label=\"batch norm\")","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:41:37.849681Z","iopub.execute_input":"2022-08-08T06:41:37.850645Z","iopub.status.idle":"2022-08-08T06:41:38.064887Z","shell.execute_reply.started":"2022-08-08T06:41:37.850598Z","shell.execute_reply":"2022-08-08T06:41:38.063536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_acc_loss(training_batchnorm, label=\"batch norm\", keyword=\"loss\")","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:41:57.858610Z","iopub.execute_input":"2022-08-08T06:41:57.859105Z","iopub.status.idle":"2022-08-08T06:41:58.071657Z","shell.execute_reply.started":"2022-08-08T06:41:57.859066Z","shell.execute_reply":"2022-08-08T06:41:58.070314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = os.listdir(\"./test1\")\ndf = pd.DataFrame(test)\n\ndf = df.rename(columns={0: \"image\"})\ndf['id'] = df['image'].apply(lambda x: x.split('.')[0])\ndf['image'] = df['image'].apply(lambda x: \"./test1/\"+x)\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:42:05.876807Z","iopub.execute_input":"2022-08-08T06:42:05.877706Z","iopub.status.idle":"2022-08-08T06:42:05.916455Z","shell.execute_reply.started":"2022-08-08T06:42:05.877661Z","shell.execute_reply":"2022-08-08T06:42:05.915513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training the model on all of input data\nmodel = train(batch_norm=True)\ntrain_on_full_data = model.fit(X, y, epochs=15)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:42:15.943587Z","iopub.execute_input":"2022-08-08T06:42:15.944882Z","iopub.status.idle":"2022-08-08T06:56:19.518143Z","shell.execute_reply.started":"2022-08-08T06:42:15.944829Z","shell.execute_reply":"2022-08-08T06:56:19.517098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_image = []\nfor i in tqdm(range(df.shape[0])):\n    img = image.load_img(df['image'][i], target_size=(64,64,1), color_mode = \"grayscale\")\n    img = image.img_to_array(img)\n    img = img/255\n    val_image.append(img)\nX_val = np.array(val_image)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:56:53.521677Z","iopub.execute_input":"2022-08-08T06:56:53.522146Z","iopub.status.idle":"2022-08-08T06:57:35.055473Z","shell.execute_reply.started":"2022-08-08T06:56:53.522110Z","shell.execute_reply":"2022-08-08T06:57:35.053979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_val)\ndf['label']=y_pred\ndf['label'] = df['label'].apply(lambda x: 0 if x<0.5 else 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:57:40.137337Z","iopub.execute_input":"2022-08-08T06:57:40.137828Z","iopub.status.idle":"2022-08-08T06:57:46.457781Z","shell.execute_reply.started":"2022-08-08T06:57:40.137781Z","shell.execute_reply":"2022-08-08T06:57:46.456559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({'id': df.id, 'label': df.label})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T06:57:50.061160Z","iopub.execute_input":"2022-08-08T06:57:50.062545Z","iopub.status.idle":"2022-08-08T06:57:50.089574Z","shell.execute_reply.started":"2022-08-08T06:57:50.062481Z","shell.execute_reply":"2022-08-08T06:57:50.088470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}