{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow import keras\nimport zipfile\nimport os\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:22:17.757607Z","iopub.execute_input":"2022-07-10T21:22:17.758073Z","iopub.status.idle":"2022-07-10T21:22:23.110351Z","shell.execute_reply.started":"2022-07-10T21:22:17.758036Z","shell.execute_reply":"2022-07-10T21:22:23.109331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with zipfile.ZipFile('../input/dogs-vs-cats/train.zip', 'r') as z:\n    z.extractall() ","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:22:23.115817Z","iopub.execute_input":"2022-07-10T21:22:23.116079Z","iopub.status.idle":"2022-07-10T21:22:34.435499Z","shell.execute_reply.started":"2022-07-10T21:22:23.116045Z","shell.execute_reply":"2022-07-10T21:22:34.434747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create two lists of filenames and labels","metadata":{}},{"cell_type":"code","source":"filenames= os.listdir('./train')\nlabels=[]\nfor name in filenames:\n    list_of_split_name= name.split('.')[0]\n    #labels.append(list_of_split_name)    \n    if list_of_split_name == 'dog':\n        labels.append('dog')\n    else:\n        labels.append('cat') ","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:22:34.439785Z","iopub.execute_input":"2022-07-10T21:22:34.441719Z","iopub.status.idle":"2022-07-10T21:22:34.515256Z","shell.execute_reply.started":"2022-07-10T21:22:34.441666Z","shell.execute_reply":"2022-07-10T21:22:34.514557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df= pd.DataFrame({ 'filename' : filenames ,'label': labels})#.sample(frac=1)\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:23:12.560106Z","iopub.execute_input":"2022-07-10T21:23:12.560368Z","iopub.status.idle":"2022-07-10T21:23:12.573154Z","shell.execute_reply.started":"2022-07-10T21:23:12.560339Z","shell.execute_reply":"2022-07-10T21:23:12.572496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:23:22.859076Z","iopub.execute_input":"2022-07-10T21:23:22.859318Z","iopub.status.idle":"2022-07-10T21:23:22.875191Z","shell.execute_reply.started":"2022-07-10T21:23:22.859291Z","shell.execute_reply":"2022-07-10T21:23:22.874292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(10) :\n    sample = filenames[i+10]\n    image = tf.keras.preprocessing.image.load_img('./train/' + sample)\n    plt.imshow(image)\n    plt.title('dog' if labels[i+10]=='dog' else 'cat')\n    plt.show()              ","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:23:24.023313Z","iopub.execute_input":"2022-07-10T21:23:24.023623Z","iopub.status.idle":"2022-07-10T21:23:27.191206Z","shell.execute_reply.started":"2022-07-10T21:23:24.023590Z","shell.execute_reply":"2022-07-10T21:23:27.190537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"split data into Train and Validation using train_test_split","metadata":{}},{"cell_type":"code","source":"train_df, valid_df = train_test_split(df , test_size= 0.3 , random_state= 42, stratify=df['label'], shuffle=True)\ntrain_df= train_df.reset_index(drop=True) \nvalid_df= valid_df.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:23:41.621799Z","iopub.execute_input":"2022-07-10T21:23:41.622054Z","iopub.status.idle":"2022-07-10T21:23:41.661608Z","shell.execute_reply.started":"2022-07-10T21:23:41.622027Z","shell.execute_reply":"2022-07-10T21:23:41.660850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ntrain_data= keras.preprocessing.image.ImageDataGenerator(rescale=1./255 ,\n                                                         rotation_range=20,\n                                                         horizontal_flip=True,\n                                                         vertical_flip=True\n                                                         )\ntrain_generator=train_data.flow_from_dataframe( dataframe=train_df,\n                                                directory='./train',\n                                                target_size=(224, 224),\n                                                x_col=\"filename\",\n                                                y_col=\"label\",\n                                                color_mode=\"rgb\",\n                                                class_mode=\"binary\",\n                                                batch_size=32,\n                                                seed = 42,\n                                                shuffle=True,\n                                                validate_filenames=True\n                                                )","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:23:43.402555Z","iopub.execute_input":"2022-07-10T21:23:43.403298Z","iopub.status.idle":"2022-07-10T21:23:43.579160Z","shell.execute_reply.started":"2022-07-10T21:23:43.403252Z","shell.execute_reply":"2022-07-10T21:23:43.578399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_data=keras.preprocessing.image.ImageDataGenerator(rescale=1./255)\n\nvalid_generator=valid_data.flow_from_dataframe( dataframe=valid_df,\n                                                directory='./train',\n                                                target_size=(224, 224),\n                                                x_col=\"filename\",\n                                                y_col=\"label\",\n                                                color_mode=\"rgb\",\n                                                class_mode=\"binary\",\n                                                batch_size=32,\n                                                seed = 42,\n                                                shuffle=True,\n                                                validate_filenames=True\n                                                )","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:23:45.326611Z","iopub.execute_input":"2022-07-10T21:23:45.327323Z","iopub.status.idle":"2022-07-10T21:23:45.411592Z","shell.execute_reply.started":"2022-07-10T21:23:45.327285Z","shell.execute_reply":"2022-07-10T21:23:45.410874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with zipfile.ZipFile('../input/dogs-vs-cats/test1.zip', 'r') as z:\n    z.extractall() ","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:23:46.595342Z","iopub.execute_input":"2022-07-10T21:23:46.595881Z","iopub.status.idle":"2022-07-10T21:23:52.592922Z","shell.execute_reply.started":"2022-07-10T21:23:46.595842Z","shell.execute_reply":"2022-07-10T21:23:52.592134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames = os.listdir(\"./test1\")\ntest_df = pd.DataFrame({'filename' : filenames})    \nsamples = test_df.shape[0]\n","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:23:52.594550Z","iopub.execute_input":"2022-07-10T21:23:52.594799Z","iopub.status.idle":"2022-07-10T21:23:52.608183Z","shell.execute_reply.started":"2022-07-10T21:23:52.594763Z","shell.execute_reply":"2022-07-10T21:23:52.607463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data=keras.preprocessing.image.ImageDataGenerator(rescale=1./255)\n\ntest_generator=train_data.flow_from_dataframe( dataframe=test_df,\n                                                directory='./test1',\n                                                target_size=(224, 224),\n                                                x_col=\"filename\",\n                                                y_col=None,\n                                                class_mode=None,\n                                                batch_size=32,\n                                                seed = 42\n                                                )","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:23:52.609671Z","iopub.execute_input":"2022-07-10T21:23:52.609965Z","iopub.status.idle":"2022-07-10T21:23:52.717144Z","shell.execute_reply.started":"2022-07-10T21:23:52.609915Z","shell.execute_reply":"2022-07-10T21:23:52.716350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import applications\n\n#base_model= keras.applications.vgg16.VGG16(weights='imagenet', include_top=False, input_shape=(224,224,3) )\nbase_model= tf.keras.applications.MobileNetV2(weights='imagenet', include_top=False, input_shape=(224,224,3) )\nbase_model.trainable= False      \n\n#base_model.summary()\n\n#drop_out1= keras.layers.Dropout(0.2)\nflatten_layer= keras.layers.Flatten()\n#drop_out2= keras.layers.Dropout(0.2)\noutput=keras.layers.Dense(1, activation= 'sigmoid')\nmodel=keras.Sequential([base_model,flatten_layer,output])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:37:28.240968Z","iopub.execute_input":"2022-07-10T21:37:28.241878Z","iopub.status.idle":"2022-07-10T21:37:29.497096Z","shell.execute_reply.started":"2022-07-10T21:37:28.241835Z","shell.execute_reply":"2022-07-10T21:37:29.495733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:37:29.958473Z","iopub.execute_input":"2022-07-10T21:37:29.959140Z","iopub.status.idle":"2022-07-10T21:37:30.028078Z","shell.execute_reply.started":"2022-07-10T21:37:29.959104Z","shell.execute_reply":"2022-07-10T21:37:30.027359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint_filepath =('./model.h5') \n\nmodel_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(filepath = checkpoint_filepath,\n                                                               monitor = 'val_loss',\n                                                               mode = 'min',\n                                                               save_best_only = True)\nearly_stopping= tf.keras.callbacks.EarlyStopping(monitor='val_accuracy',\n                                                 mode='max',\n                                                 patience=5,\n                                                 restore_best_weights= True,\n                                                 )","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:37:30.904485Z","iopub.execute_input":"2022-07-10T21:37:30.904843Z","iopub.status.idle":"2022-07-10T21:37:30.910599Z","shell.execute_reply.started":"2022-07-10T21:37:30.904802Z","shell.execute_reply":"2022-07-10T21:37:30.909621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam', loss=tf.keras.losses.BinaryCrossentropy(), metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:37:40.453010Z","iopub.execute_input":"2022-07-10T21:37:40.453256Z","iopub.status.idle":"2022-07-10T21:37:40.468109Z","shell.execute_reply.started":"2022-07-10T21:37:40.453228Z","shell.execute_reply":"2022-07-10T21:37:40.467346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history= model.fit_generator(train_generator, steps_per_epoch = train_generator.samples // 32,validation_data=valid_generator,validation_steps = valid_generator.samples // 32 ,epochs=20,callbacks=[model_checkpoint_callback, early_stopping] )","metadata":{"execution":{"iopub.status.busy":"2022-07-10T21:37:41.857492Z","iopub.execute_input":"2022-07-10T21:37:41.857759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\n\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T23:10:44.220388Z","iopub.execute_input":"2022-07-04T23:10:44.220691Z","iopub.status.idle":"2022-07-04T23:10:44.53507Z","shell.execute_reply.started":"2022-07-04T23:10:44.220639Z","shell.execute_reply":"2022-07-04T23:10:44.534046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predict = model.predict_generator(test_generator, steps=np.ceil(samples/32))","metadata":{"execution":{"iopub.status.busy":"2022-07-04T23:10:44.539491Z","iopub.execute_input":"2022-07-04T23:10:44.539858Z","iopub.status.idle":"2022-07-04T23:13:38.081811Z","shell.execute_reply.started":"2022-07-04T23:10:44.539815Z","shell.execute_reply":"2022-07-04T23:13:38.080843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['category'] = np.argmax(predict, axis=-1)\ntest_df['category'] = test_df['category'].replace({ 'dog': 1, 'cat': 0 })","metadata":{"execution":{"iopub.status.busy":"2022-07-04T23:13:38.083606Z","iopub.execute_input":"2022-07-04T23:13:38.084133Z","iopub.status.idle":"2022-07-04T23:13:38.091294Z","shell.execute_reply.started":"2022-07-04T23:13:38.084093Z","shell.execute_reply":"2022-07-04T23:13:38.090623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = test_df.copy()\nsubmission_df['id'] = submission_df['filename'].str.split('.').str[0]\nsubmission_df['label'] = submission_df['category']\nsubmission_df.drop(['filename', 'category'], axis=1, inplace=True)\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T23:13:38.092725Z","iopub.execute_input":"2022-07-04T23:13:38.093351Z","iopub.status.idle":"2022-07-04T23:13:38.161292Z","shell.execute_reply.started":"2022-07-04T23:13:38.09331Z","shell.execute_reply":"2022-07-04T23:13:38.160376Z"},"trusted":true},"execution_count":null,"outputs":[]}]}