{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this project, we will see how to use Keras and TensorFlow to build, train, and test a Convolutional Neural Network capable of identifying the breed of a dog in a supplied image. This is a **supervised learning** problem, specifically **a multiclass classification** problem.","metadata":{}},{"cell_type":"code","source":"import numpy as np          # linear algebra\nimport pandas as pd        # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom random import randint\nfrom tqdm import tqdm      #tqdm is a library in Python which is used for creating Progress Meters or Progress Bars. \n#Look at this: https://www.analyticsvidhya.com/blog/2021/05/how-to-use-progress-bars-in-python/\nimport cv2\nimport matplotlib.pyplot as plt\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nfrom subprocess import check_output\nprint(check_output([\"ls\", \"../input\"]).decode(\"utf8\"))\n\n# Any results you write to the current directory are saved as output.","metadata":{"_uuid":"ae0df6a520487e080515d7a76cf955d095b483c0","_cell_guid":"499ebdfb-fede-4ba2-9ff4-067292f9a865","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing import image\nfrom sklearn.preprocessing import label_binarize\nfrom sklearn.model_selection import train_test_split\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D\nfrom keras.optimizers import Adam\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading the labels data into dataframe and viewing it. Here we analysed that labels contains 10222 rows and 2 columns.  ","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/labels.csv\")\nprint(df_train.shape)\ndf_train.head(10)\n","metadata":{"_uuid":"8a9e1edb957ac7a7706a2248902f87ca8b39bc0f","_cell_guid":"1a219277-cf22-4d2d-84cf-3b421f591913","execution":{"iopub.status.busy":"2022-07-14T15:52:55.310964Z","iopub.execute_input":"2022-07-14T15:52:55.311560Z","iopub.status.idle":"2022-07-14T15:52:55.382294Z","shell.execute_reply.started":"2022-07-14T15:52:55.311501Z","shell.execute_reply":"2022-07-14T15:52:55.381750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Visualize trainingsdata distribution***","metadata":{"_uuid":"d41536607ab5486cba4beaf9f0e93798065af702","_cell_guid":"52de0fd9-a626-4980-a474-7ae7d00b9a30"}},{"cell_type":"code","source":"# Visualize the number of each breeds\nbreeds_all = df_train[\"breed\"]\nbreed_counts = breeds_all.value_counts()\nbreed_counts.head()\n\n#Here we are finding out the count per class i.e. total data in each class using value_counts() function.\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:53:03.166727Z","iopub.execute_input":"2022-07-14T15:53:03.167281Z","iopub.status.idle":"2022-07-14T15:53:03.179558Z","shell.execute_reply.started":"2022-07-14T15:53:03.167232Z","shell.execute_reply":"2022-07-14T15:53:03.178839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_images(images, classes):\n    assert len(images) == len(classes) == 9\n    \n    # Create figure with 3x3 sub-plots.\n    fig, axes = plt.subplots(3, 3,figsize=(60,60),sharex=True)\n    fig.subplots_adjust(hspace=0.3, wspace=0.3)\n   \n    for i, ax in enumerate(axes.flat):\n        # Plot image.\n        \n        ax.imshow(cv2.cvtColor(images[i], cv2.COLOR_BGR2RGB).reshape(img_width,img_height,3), cmap='hsv')    \n        xlabel = \"Breed: {0}\".format(classes[i])\n    \n        # Show the classes as the label on the x-axis.\n        ax.set_xlabel(xlabel)\n        ax.xaxis.label.set_size(60)\n        # Remove ticks from the plot.\n        ax.set_xticks([])\n        ax.set_yticks([])\n    \n    # Ensure the plot is shown correctly with multiple plots\n    # in a single Notebook cell.\n    \n    plt.show()","metadata":{"_uuid":"2beabc5637359f977b784e607a33bcfb79e44193","_cell_guid":"ce5b56e1-f9d3-4d5a-9dac-8567921d26e9","execution":{"iopub.status.busy":"2022-07-14T15:53:10.683203Z","iopub.execute_input":"2022-07-14T15:53:10.683844Z","iopub.status.idle":"2022-07-14T15:53:10.698532Z","shell.execute_reply.started":"2022-07-14T15:53:10.683791Z","shell.execute_reply":"2022-07-14T15:53:10.697730Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Load trainingsdata***","metadata":{"_uuid":"e9a23a1858b115b6260c0bb419234db8d4741d87","_cell_guid":"a280db79-ce8f-4742-9ee4-599f1769f397"}},{"cell_type":"code","source":"img_width=250\nimg_height=250\nimages=[]\nclasses=[]\n#load training images\nfor f, breed in tqdm(df_train.values):\n    img = cv2.imread('../input/train/{}.jpg'.format(f))\n    classes.append(breed)\n    images.append(cv2.resize(img, (img_width, img_height)))","metadata":{"_uuid":"f41baaf90b16b8957a86a847aa98f723c520655d","_cell_guid":"a55f656a-1af1-4ee1-814d-86f984692f7f","execution":{"iopub.status.busy":"2022-07-14T15:53:16.382216Z","iopub.execute_input":"2022-07-14T15:53:16.382814Z","iopub.status.idle":"2022-07-14T15:55:49.109985Z","shell.execute_reply.started":"2022-07-14T15:53:16.382760Z","shell.execute_reply":"2022-07-14T15:55:49.109234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Plot some example images***","metadata":{"_uuid":"ae67ce24433159dad5f3ea5d16d0e1c3e504932c","_cell_guid":"78666aef-0728-4dd0-a601-2792fac909ba"}},{"cell_type":"code","source":"\n# select random images\nrandom_numbers = [randint(0, len(images)) for p in range(0,9)]\nprint(random_numbers)\nimages_to_show = [images[i] for i in random_numbers]\nclasses_to_show = [classes[i] for i in random_numbers]\nprint(\"Images to show: {0}\".format(len(images_to_show)))\nprint(\"Classes to show: {0}\".format(len(classes_to_show)))\n\n#plot the images\nplot_images(images_to_show, classes_to_show)\n   ","metadata":{"_uuid":"f9c8f2ac1b6feab999464e3f64c90bf49e8e25b1","_cell_guid":"8bd8db60-26f5-4b28-92e8-6ae1f34b124e","execution":{"iopub.status.busy":"2022-07-14T15:57:48.731531Z","iopub.execute_input":"2022-07-14T15:57:48.731858Z","iopub.status.idle":"2022-07-14T15:57:51.865139Z","shell.execute_reply.started":"2022-07-14T15:57:48.731819Z","shell.execute_reply":"2022-07-14T15:57:51.864463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Selecting first 3 breeds (Limitation due to computation power)\nCLASS_NAMES = ['scottish_deerhound','maltese_dog','bernese_mountain_dog']\nlabels = df_train[(df_train['breed'].isin(CLASS_NAMES))]\nlabels = labels.reset_index()\nlabels.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:58:00.298273Z","iopub.execute_input":"2022-07-14T15:58:00.298586Z","iopub.status.idle":"2022-07-14T15:58:00.316360Z","shell.execute_reply.started":"2022-07-14T15:58:00.298548Z","shell.execute_reply":"2022-07-14T15:58:00.315454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating numpy matrix with zeros\nX_data = np.zeros((len(labels), 224, 224, 3), dtype='float32')\n# One hot encoding\nY_data = label_binarize(labels['breed'], classes = CLASS_NAMES)\n\n# Reading and converting image to numpy array and normalizing dataset\nfor i in tqdm(range(len(labels))):\n    img = image.load_img('../input/train/%s.jpg' % labels['id'][i], target_size=(224, 224))\n    img = image.img_to_array(img)\n    x = np.expand_dims(img.copy(), axis=0)\n    X_data[i] = x / 255.0\n    \n# Printing train image and one hot encode shape & size\nprint('\\nTrain Images shape: ',X_data.shape,' size: {:,}'.format(X_data.size))\nprint('One-hot encoded output shape: ',Y_data.shape,' size: {:,}'.format(Y_data.size))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:58:07.091953Z","iopub.execute_input":"2022-07-14T15:58:07.092521Z","iopub.status.idle":"2022-07-14T15:58:09.128729Z","shell.execute_reply.started":"2022-07-14T15:58:07.092473Z","shell.execute_reply":"2022-07-14T15:58:09.127868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we are working with the classification dataset first we need to one hot encode the target value i.e. the classes. After that we will read images and convert them into numpy array and finally normalizing the array.","metadata":{}},{"cell_type":"code","source":"# Building the Model\nmodel = Sequential()\n\nmodel.add(Conv2D(filters = 64, kernel_size = (5,5), activation ='relu', input_shape = (224,224,3)))\nmodel.add(MaxPool2D(pool_size=(2,2)))\n\nmodel.add(Conv2D(filters = 32, kernel_size = (3,3), activation ='relu', kernel_regularizer = 'l2'))\nmodel.add(MaxPool2D(pool_size=(2,2)))\n\nmodel.add(Conv2D(filters = 16, kernel_size = (7,7), activation ='relu', kernel_regularizer = 'l2'))\nmodel.add(MaxPool2D(pool_size=(2,2)))\n\nmodel.add(Conv2D(filters = 8, kernel_size = (5,5), activation ='relu', kernel_regularizer = 'l2'))\nmodel.add(MaxPool2D(pool_size=(2,2)))\n\nmodel.add(Flatten())\nmodel.add(Dense(128, activation = \"relu\", kernel_regularizer = 'l2'))\nmodel.add(Dense(64, activation = \"relu\", kernel_regularizer = 'l2'))\nmodel.add(Dense(len(CLASS_NAMES), activation = \"softmax\"))\n\nmodel.compile(loss = 'categorical_crossentropy', optimizer = Adam(0.0001),metrics=['accuracy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:58:15.891878Z","iopub.execute_input":"2022-07-14T15:58:15.892529Z","iopub.status.idle":"2022-07-14T15:58:16.098911Z","shell.execute_reply.started":"2022-07-14T15:58:15.892470Z","shell.execute_reply":"2022-07-14T15:58:16.098211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next we will create a network architecture for the model. We have used different types of layers according to their features namely Conv_2d (It is used to create a convolutional kernel that is convolved with the input layer to produce the output tensor), max_pooling2d (It is a downsampling technique which takes out the maximum value over the window defined by poolsize), flatten (It flattens the input and creates a 1D output), Dense (Dense layer produce the output as the dot product of input and kernel).\n","metadata":{}},{"cell_type":"markdown","source":"After defining the network architecture we found out the total parameters as 162,619.\n","metadata":{}},{"cell_type":"code","source":"# Splitting the data set into training and testing data sets\nX_train_and_val, X_test, Y_train_and_val, Y_test = train_test_split(X_data, Y_data, test_size = 0.1)\n# Splitting the training data set into training and validation data sets\nX_train, X_val, Y_train, Y_val = train_test_split(X_train_and_val, Y_train_and_val, test_size = 0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:58:22.605185Z","iopub.execute_input":"2022-07-14T15:58:22.605840Z","iopub.status.idle":"2022-07-14T15:58:22.971619Z","shell.execute_reply.started":"2022-07-14T15:58:22.605787Z","shell.execute_reply":"2022-07-14T15:58:22.970843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After defining the network architecture we will start with splitting the test and train data then dividing train data in train and validation data. \n","metadata":{}},{"cell_type":"code","source":"# Training the model\nepochs = 100\nbatch_size = 128\n\nhistory = model.fit(X_train, Y_train, batch_size = batch_size, epochs = epochs, \n                    validation_data = (X_val, Y_val))","metadata":{"execution":{"iopub.status.busy":"2022-07-14T15:58:27.251186Z","iopub.execute_input":"2022-07-14T15:58:27.251831Z","iopub.status.idle":"2022-07-14T17:17:34.117146Z","shell.execute_reply.started":"2022-07-14T15:58:27.251778Z","shell.execute_reply":"2022-07-14T17:17:34.115684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we will train our model on 100 epochs and a batch size of 128. You can try using more number of epochs to increase accuracy. During each epochs we can see how the model is performing by viewing the training and validation accuracy.\n","metadata":{}},{"cell_type":"code","source":"# Plot the training history\nplt.figure(figsize=(12, 5))\nplt.plot(history.history['acc'], color='r')\nplt.plot(history.history['val_acc'], color='b')\nplt.title('Model Accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epochs')\nplt.legend(['train', 'val'])\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T17:18:25.117779Z","iopub.execute_input":"2022-07-14T17:18:25.118181Z","iopub.status.idle":"2022-07-14T17:18:25.328491Z","shell.execute_reply.started":"2022-07-14T17:18:25.118114Z","shell.execute_reply":"2022-07-14T17:18:25.327673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we analyse how the model is learning with each epoch in terms of accuracy.","metadata":{}},{"cell_type":"code","source":"Y_pred = model.predict(X_test)\nscore = model.evaluate(X_test, Y_test)\nprint('Accuracy over the test set: \\n ', round((score[1]*100), 2), '%')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T17:18:35.672234Z","iopub.execute_input":"2022-07-14T17:18:35.672859Z","iopub.status.idle":"2022-07-14T17:18:40.516082Z","shell.execute_reply.started":"2022-07-14T17:18:35.672807Z","shell.execute_reply":"2022-07-14T17:18:40.515225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will use predict function to make predictions using this model also we are finding out the accuracy on the test set.\n","metadata":{}},{"cell_type":"code","source":"# Plotting image to compare\nplt.imshow(X_test[1,:,:,:])\nplt.show()\n\n# Finding max value from predition list and comaparing original value vs predicted\nprint(\"Originally : \",labels['breed'][np.argmax(Y_test[1])])\nprint(\"Predicted : \",labels['breed'][np.argmax(Y_pred[1])])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T17:18:52.715494Z","iopub.execute_input":"2022-07-14T17:18:52.715779Z","iopub.status.idle":"2022-07-14T17:18:52.884237Z","shell.execute_reply.started":"2022-07-14T17:18:52.715743Z","shell.execute_reply":"2022-07-14T17:18:52.883562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Well Done!**","metadata":{"execution":{"iopub.status.busy":"2022-02-17T09:53:42.284607Z","iopub.execute_input":"2022-02-17T09:53:42.284916Z","iopub.status.idle":"2022-02-17T09:53:42.290025Z","shell.execute_reply.started":"2022-02-17T09:53:42.284879Z","shell.execute_reply":"2022-02-17T09:53:42.288772Z"}}}]}