{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# This kernel is inspire from these two kernels:\n\nhttps://www.kaggle.com/code/uysimty/keras-cnn-dog-or-cat-classification/notebook\n\nhttps://www.kaggle.com/code/yassineghouzam/introduction-to-cnn-keras-0-997-top-6/notebook\n\nThis is the first kenel for me using CNN with image data.So, I make a combination of the above kernels. This is not a copy, I make changes with my knowledge and combine these kernels and i'm in progress.","metadata":{}},{"cell_type":"markdown","source":"# Pls upvote if you like this kernel.","metadata":{}},{"cell_type":"code","source":"#import libraries\nimport numpy as np#linear algebra\nimport pandas as pd#data processing\nimport seaborn as sns\nimport matplotlib.pyplot as plt#visualization\nimport matplotlib.image as mpimg\n\nimport warnings\nwarnings.filterwarnings('ignore')#to ignore warnings\n\nimport zipfile#for unzipping\n\nfrom keras.preprocessing.image import ImageDataGenerator, load_img\nfrom keras.utils.np_utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\n\nimport random\nimport os\nprint(os.listdir(\"../input\"))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-16T07:08:22.384792Z","iopub.execute_input":"2022-07-16T07:08:22.385275Z","iopub.status.idle":"2022-07-16T07:08:22.395272Z","shell.execute_reply.started":"2022-07-16T07:08:22.385235Z","shell.execute_reply":"2022-07-16T07:08:22.394252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dfine Constants\nwe have to set the input shape of images that fit to the model with (128*128 pixels and 3 channels, R,G,B)","metadata":{}},{"cell_type":"code","source":"#define constants\nFAST_RUN = False\nIMAGE_WIDTH = 128\nIMAGE_HEIGHT = 128\nIMAGE_SIZE = (IMAGE_WIDTH, IMAGE_HEIGHT)\nIMAGE_CHANNELS = 3\n","metadata":{"execution":{"iopub.status.busy":"2022-07-16T00:26:56.153355Z","iopub.execute_input":"2022-07-16T00:26:56.153805Z","iopub.status.idle":"2022-07-16T00:26:56.159358Z","shell.execute_reply.started":"2022-07-16T00:26:56.153768Z","shell.execute_reply":"2022-07-16T00:26:56.158241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Unzip data and remake a new data frame for training set\n","metadata":{}},{"cell_type":"code","source":"# prepare data\nwith zipfile.ZipFile(\"../input/dogs-vs-cats-redux-kernels-edition/train.zip\",\"r\") as z:\n    z.extractall(\".\")\n\nwith zipfile.ZipFile(\"../input/dogs-vs-cats-redux-kernels-edition/test.zip\",\"r\") as z:\n    z.extractall(\".\")\n\ntrain_path = \"/kaggle/working/train\"\nfilenames = os.listdir(train_path)\n#set the dog to 1 and cat to 0 \ncategories = []\nfor filename in filenames:\n    category = filename.split(\".\")[0]\n    if category == 'dog':\n        categories.append(1)\n    else:\n        categories.append(0)\ndf = pd.DataFrame({\n    'filename':filenames,\n    'category':categories\n})","metadata":{"execution":{"iopub.status.busy":"2022-07-16T00:26:56.189309Z","iopub.execute_input":"2022-07-16T00:26:56.189962Z","iopub.status.idle":"2022-07-16T00:27:10.670163Z","shell.execute_reply.started":"2022-07-16T00:26:56.189924Z","shell.execute_reply":"2022-07-16T00:27:10.669010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing data","metadata":{}},{"cell_type":"code","source":"print(df.head(),\"\\n--------\\n\",df.tail())","metadata":{"execution":{"iopub.status.busy":"2022-07-16T00:27:10.672078Z","iopub.execute_input":"2022-07-16T00:27:10.672500Z","iopub.status.idle":"2022-07-16T00:27:10.681467Z","shell.execute_reply.started":"2022-07-16T00:27:10.672463Z","shell.execute_reply":"2022-07-16T00:27:10.680433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#data distribution (classes)\ng = sns.countplot(df['category'])\ndf['category'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T00:27:10.683009Z","iopub.execute_input":"2022-07-16T00:27:10.683358Z","iopub.status.idle":"2022-07-16T00:27:10.805486Z","shell.execute_reply.started":"2022-07-16T00:27:10.683326Z","shell.execute_reply":"2022-07-16T00:27:10.803339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Checking NUll  ","metadata":{}},{"cell_type":"code","source":"df['filename'].isnull().any()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T00:27:10.808839Z","iopub.execute_input":"2022-07-16T00:27:10.809731Z","iopub.status.idle":"2022-07-16T00:27:10.820405Z","shell.execute_reply.started":"2022-07-16T00:27:10.809695Z","shell.execute_reply":"2022-07-16T00:27:10.819211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['category'].isnull().any()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T00:27:10.822395Z","iopub.execute_input":"2022-07-16T00:27:10.823255Z","iopub.status.idle":"2022-07-16T00:27:10.839967Z","shell.execute_reply.started":"2022-07-16T00:27:10.823210Z","shell.execute_reply":"2022-07-16T00:27:10.838696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# See some input smaple","metadata":{}},{"cell_type":"code","source":"print(df['filename'][0])","metadata":{"execution":{"iopub.status.busy":"2022-07-16T00:27:10.842474Z","iopub.execute_input":"2022-07-16T00:27:10.843270Z","iopub.status.idle":"2022-07-16T00:27:10.849871Z","shell.execute_reply.started":"2022-07-16T00:27:10.843224Z","shell.execute_reply":"2022-07-16T00:27:10.848652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = mpimg.imread('/kaggle/working/train/dog.3623.jpg')\nimgplot = plt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T00:27:10.852088Z","iopub.execute_input":"2022-07-16T00:27:10.852981Z","iopub.status.idle":"2022-07-16T00:27:11.060475Z","shell.execute_reply.started":"2022-07-16T00:27:10.852934Z","shell.execute_reply":"2022-07-16T00:27:11.059219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making a CNN model refernece from\n\nhttps://www.kaggle.com/code/uysimty/keras-cnn-dog-or-cat-classification/notebook","metadata":{}},{"cell_type":"code","source":"\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Dropout, Flatten, Dense, Activation, BatchNormalization\n\nmodel = Sequential()\n\nmodel.add(Conv2D(32, (3, 3), activation='relu', input_shape=(IMAGE_WIDTH, IMAGE_HEIGHT, IMAGE_CHANNELS)))\nmodel.add(BatchNormalization())\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Conv2D(64, (3, 3), activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Conv2D(128, (3, 3), activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(MaxPooling2D(pool_size=(2, 2)))\nmodel.add(Dropout(0.25))\n\nmodel.add(Flatten())\nmodel.add(Dense(512, activation='relu'))\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.5))\nmodel.add(Dense(2, activation='softmax')) # 2 because we have cat and dog classes\n\nmodel.compile(loss='categorical_crossentropy', optimizer='rmsprop', metrics=['accuracy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T02:01:44.757874Z","iopub.execute_input":"2022-07-16T02:01:44.758341Z","iopub.status.idle":"2022-07-16T02:01:44.956795Z","shell.execute_reply.started":"2022-07-16T02:01:44.758306Z","shell.execute_reply":"2022-07-16T02:01:44.955681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training will stop if it is not improved druing 10 epochs\n# Learning rate will be reduced, if training is not improved after 2 epochs","metadata":{}},{"cell_type":"code","source":"from keras.callbacks import EarlyStopping, ReduceLROnPlateau\nearlystop = EarlyStopping(patience=10)\nlearning_rate_reduction = ReduceLROnPlateau(monitor=\"val_loss\", \n                                            patience=2, \n                                            verbose=1, \n                                            factor=0.5, \n                                            min_lr=0.00001)\ncallbacks = [earlystop, learning_rate_reduction]","metadata":{"execution":{"iopub.status.busy":"2022-07-16T02:01:56.313818Z","iopub.execute_input":"2022-07-16T02:01:56.314172Z","iopub.status.idle":"2022-07-16T02:01:56.320840Z","shell.execute_reply.started":"2022-07-16T02:01:56.314143Z","shell.execute_reply":"2022-07-16T02:01:56.319634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# set the training size to 90% and validation to 10%","metadata":{}},{"cell_type":"code","source":"df['category']=df['category'].replace({0:'cat', 1:'dog'})\ntrain_df, validate_df = train_test_split(df, test_size=0.1, random_state=42)\ntrain_df = train_df.reset_index(drop=True)\nvalidate_df = validate_df.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T02:01:58.432157Z","iopub.execute_input":"2022-07-16T02:01:58.432558Z","iopub.status.idle":"2022-07-16T02:01:58.450796Z","shell.execute_reply.started":"2022-07-16T02:01:58.432524Z","shell.execute_reply":"2022-07-16T02:01:58.450014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualizing the train and validation data","metadata":{}},{"cell_type":"code","source":"g = sns.countplot(train_df['category'])\ntrain_df['category'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T02:02:03.646124Z","iopub.execute_input":"2022-07-16T02:02:03.646500Z","iopub.status.idle":"2022-07-16T02:02:03.842877Z","shell.execute_reply.started":"2022-07-16T02:02:03.646469Z","shell.execute_reply":"2022-07-16T02:02:03.841624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"g = sns.countplot(validate_df['category'])\nvalidate_df['category'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T02:02:04.934285Z","iopub.execute_input":"2022-07-16T02:02:04.934713Z","iopub.status.idle":"2022-07-16T02:02:05.091855Z","shell.execute_reply.started":"2022-07-16T02:02:04.934679Z","shell.execute_reply":"2022-07-16T02:02:05.090822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Setting batch_size and epochs; Note: set the epochs to 30 to get 91% accuracy","metadata":{}},{"cell_type":"code","source":"total_train = train_df.shape[0]\ntotal_validate = validate_df.shape[0]\nbatch_size=15\nepochs =1","metadata":{"execution":{"iopub.status.busy":"2022-07-16T02:02:07.414671Z","iopub.execute_input":"2022-07-16T02:02:07.415756Z","iopub.status.idle":"2022-07-16T02:02:07.420443Z","shell.execute_reply.started":"2022-07-16T02:02:07.415706Z","shell.execute_reply":"2022-07-16T02:02:07.419649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# To avoid overfitting, generate new data for training and validation  ","metadata":{}},{"cell_type":"code","source":"datagen = ImageDataGenerator(\n        featurewise_center=False,  # set input mean to 0 over the dataset\n        samplewise_center=False,  # set each sample mean to 0\n        featurewise_std_normalization=False,  # divide inputs by std of the dataset\n        samplewise_std_normalization=False,  # divide each input by its std\n        zca_whitening=False,  # apply ZCA whitening\n        rotation_range=10,  # randomly rotate images in the range (degrees, 0 to 180)\n        rescale = 1./255,\n        zoom_range = 0.1, # Randomly zoom image \n        width_shift_range=0.1,  # randomly shift images horizontally (fraction of total width)\n        height_shift_range=0.1,  # randomly shift images vertically (fraction of total height)\n        horizontal_flip=False,  # randomly flip images\n        vertical_flip=False)  # randomly flip images\n\ntrain_generator = datagen.flow_from_dataframe(\n                train_df,\n                \"/kaggle/working/train/\",\n                x_col='filename',\n                y_col='category',\n                target_size=IMAGE_SIZE,\n                class_mode='categorical',\n                batch_size=batch_size\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T02:02:09.585595Z","iopub.execute_input":"2022-07-16T02:02:09.586288Z","iopub.status.idle":"2022-07-16T02:02:09.881407Z","shell.execute_reply.started":"2022-07-16T02:02:09.586252Z","shell.execute_reply":"2022-07-16T02:02:09.879444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"validation_datagen = ImageDataGenerator(rescale=1./255)\nvalidation_generator = validation_datagen.flow_from_dataframe(\n    validate_df, \n    \"/kaggle/working/train/\", \n    x_col='filename',\n    y_col='category',\n    target_size=IMAGE_SIZE,\n    class_mode='categorical',\n    batch_size=batch_size\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T02:02:11.073788Z","iopub.execute_input":"2022-07-16T02:02:11.074173Z","iopub.status.idle":"2022-07-16T02:02:11.115471Z","shell.execute_reply.started":"2022-07-16T02:02:11.074142Z","shell.execute_reply":"2022-07-16T02:02:11.114267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualize some example from generator","metadata":{}},{"cell_type":"code","source":"example_df = train_df.sample(n=1).reset_index(drop=True)\nexample_generator = datagen.flow_from_dataframe(\n    example_df, \n    \"/kaggle/working/train/\", \n    x_col='filename',\n    y_col='category',\n    target_size=IMAGE_SIZE,\n    class_mode='categorical'\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T02:02:11.870410Z","iopub.execute_input":"2022-07-16T02:02:11.871025Z","iopub.status.idle":"2022-07-16T02:02:11.881756Z","shell.execute_reply.started":"2022-07-16T02:02:11.870988Z","shell.execute_reply":"2022-07-16T02:02:11.880955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 12))\nfor i in range(0, 15):\n    plt.subplot(5, 3, i+1)\n    for X_batch, Y_batch in example_generator:\n        image = X_batch[0]\n        plt.imshow(image)\n        break\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T02:02:12.661269Z","iopub.execute_input":"2022-07-16T02:02:12.661914Z","iopub.status.idle":"2022-07-16T02:02:14.554028Z","shell.execute_reply.started":"2022-07-16T02:02:12.661877Z","shell.execute_reply":"2022-07-16T02:02:14.553252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"history = model.fit_generator(\n    train_generator, \n    epochs=epochs,\n    validation_data=validation_generator,\n    validation_steps=total_validate//batch_size,\n    steps_per_epoch=total_train//batch_size,\n    callbacks=callbacks\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T02:02:16.783581Z","iopub.execute_input":"2022-07-16T02:02:16.784418Z","iopub.status.idle":"2022-07-16T06:04:22.482397Z","shell.execute_reply.started":"2022-07-16T02:02:16.784342Z","shell.execute_reply":"2022-07-16T06:04:22.480243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save the parameters of the model","metadata":{}},{"cell_type":"code","source":"model.save_weights(\"model.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-07-16T06:05:54.539836Z","iopub.execute_input":"2022-07-16T06:05:54.540344Z","iopub.status.idle":"2022-07-16T06:05:54.766186Z","shell.execute_reply.started":"2022-07-16T06:05:54.540267Z","shell.execute_reply":"2022-07-16T06:05:54.765040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analyze the accuracy and loss","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots(2,1, figsize=(12,12))\nax[0].plot(history.history['loss'], color='b', label='Training Loss')\nax[0].plot(history.history['val_loss'], color='r', label='Validation Loss')\nlegend = ax[0].legend(loc='best', shadow=True)\n\nax[1].plot(history.history['accuracy'], color='g', label='Training Accuracy')\nax[1].plot(history.history['val_accuracy'], color='y', label='Validation Accuracy')\nlegend = ax[1].legend(loc='best', shadow=True)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-16T06:25:35.949813Z","iopub.execute_input":"2022-07-16T06:25:35.951132Z","iopub.status.idle":"2022-07-16T06:25:36.305419Z","shell.execute_reply.started":"2022-07-16T06:25:35.951081Z","shell.execute_reply":"2022-07-16T06:25:36.304323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare testing data","metadata":{}},{"cell_type":"code","source":"test_filenames = os.listdir(\"/kaggle/working/test\")\ntest_df = pd.DataFrame({\n          'filename':test_filenames\n    \n})\nnp_samples=test_df.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-16T06:40:29.462342Z","iopub.execute_input":"2022-07-16T06:40:29.463604Z","iopub.status.idle":"2022-07-16T06:40:29.479445Z","shell.execute_reply.started":"2022-07-16T06:40:29.463560Z","shell.execute_reply":"2022-07-16T06:40:29.478591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T06:40:30.554998Z","iopub.execute_input":"2022-07-16T06:40:30.555667Z","iopub.status.idle":"2022-07-16T06:40:30.560908Z","shell.execute_reply.started":"2022-07-16T06:40:30.555629Z","shell.execute_reply":"2022-07-16T06:40:30.559750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_df.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-16T06:40:31.401487Z","iopub.execute_input":"2022-07-16T06:40:31.401879Z","iopub.status.idle":"2022-07-16T06:40:31.407274Z","shell.execute_reply.started":"2022-07-16T06:40:31.401847Z","shell.execute_reply":"2022-07-16T06:40:31.405642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate the test data for model evaluation","metadata":{}},{"cell_type":"code","source":"test_gen = ImageDataGenerator(rescale=1./255)\ntest_generator = test_gen.flow_from_dataframe(\n    test_df, \n    \"/kaggle/working/test\", \n    x_col='filename',\n    y_col=None,\n    class_mode=None,\n    target_size=IMAGE_SIZE,\n    batch_size=batch_size,\n    shuffle=False\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T06:40:33.945168Z","iopub.execute_input":"2022-07-16T06:40:33.945868Z","iopub.status.idle":"2022-07-16T06:40:34.076551Z","shell.execute_reply.started":"2022-07-16T06:40:33.945815Z","shell.execute_reply":"2022-07-16T06:40:34.075359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making prediction","metadata":{}},{"cell_type":"code","source":"predict = model.predict_generator(test_generator, steps=np.ceil(np_samples/batch_size))","metadata":{"execution":{"iopub.status.busy":"2022-07-16T06:40:43.595776Z","iopub.execute_input":"2022-07-16T06:40:43.596550Z","iopub.status.idle":"2022-07-16T06:42:05.835907Z","shell.execute_reply.started":"2022-07-16T06:40:43.596501Z","shell.execute_reply":"2022-07-16T06:42:05.834798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T06:49:06.121997Z","iopub.execute_input":"2022-07-16T06:49:06.122478Z","iopub.status.idle":"2022-07-16T06:49:06.132503Z","shell.execute_reply.started":"2022-07-16T06:49:06.122443Z","shell.execute_reply":"2022-07-16T06:49:06.130848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['category'] = np.argmax(predict, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T06:49:58.498418Z","iopub.execute_input":"2022-07-16T06:49:58.498879Z","iopub.status.idle":"2022-07-16T06:49:58.506588Z","shell.execute_reply.started":"2022-07-16T06:49:58.498841Z","shell.execute_reply":"2022-07-16T06:49:58.505502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_map = dict((v,k) for k,v in train_generator.class_indices.items())\ntest_df['category'] = test_df['category'].replace(label_map)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T06:55:14.128232Z","iopub.execute_input":"2022-07-16T06:55:14.128667Z","iopub.status.idle":"2022-07-16T06:55:14.138172Z","shell.execute_reply.started":"2022-07-16T06:55:14.128632Z","shell.execute_reply":"2022-07-16T06:55:14.137191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['category'] = test_df['category'].replace({ 'dog': 1, 'cat': 0 })","metadata":{"execution":{"iopub.status.busy":"2022-07-16T06:55:35.860363Z","iopub.execute_input":"2022-07-16T06:55:35.861068Z","iopub.status.idle":"2022-07-16T06:55:35.876012Z","shell.execute_reply.started":"2022-07-16T06:55:35.861030Z","shell.execute_reply":"2022-07-16T06:55:35.874891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['category'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T06:55:43.939640Z","iopub.execute_input":"2022-07-16T06:55:43.940049Z","iopub.status.idle":"2022-07-16T06:55:44.126448Z","shell.execute_reply.started":"2022-07-16T06:55:43.940012Z","shell.execute_reply":"2022-07-16T06:55:44.125338Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# See predicted results","metadata":{}},{"cell_type":"code","source":"sample_test = test_df.head(18)\nsample_test.head()\nplt.figure(figsize=(12, 24))\nfor index, row in sample_test.iterrows():\n    filename = row['filename']\n    category = row['category']\n    img = load_img(\"/kaggle/working/test/\"+filename, target_size=IMAGE_SIZE)\n    plt.subplot(6, 3, index+1)\n    plt.imshow(img)\n    plt.xlabel(filename + '(' + \"{}\".format(category) + ')' )\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:00:18.042532Z","iopub.execute_input":"2022-07-16T07:00:18.043190Z","iopub.status.idle":"2022-07-16T07:00:21.211053Z","shell.execute_reply.started":"2022-07-16T07:00:18.043153Z","shell.execute_reply":"2022-07-16T07:00:21.210161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission_df = test_df.copy()\nsubmission_df['id'] = submission_df['filename'].str.split('.').str[0]\nsubmission_df['label'] = submission_df['category']\nsubmission_df.drop(['filename', 'category'], axis=1, inplace=True)\nsubmission_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-16T07:01:00.592538Z","iopub.execute_input":"2022-07-16T07:01:00.592930Z","iopub.status.idle":"2022-07-16T07:01:00.985184Z","shell.execute_reply.started":"2022-07-16T07:01:00.592900Z","shell.execute_reply":"2022-07-16T07:01:00.984224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}