{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Happywhale - Whale and Dolphin Identification","metadata":{}},{"cell_type":"markdown","source":"![](https://storage.googleapis.com/kaggle-media/competitions/Happywhale/AU%20Kaggle%20Competition%20Description%20Image-03.jpg)","metadata":{}},{"cell_type":"markdown","source":"Importing Library","metadata":{}},{"cell_type":"code","source":"# These library are for data manipulation \nimport numpy as np\nimport pandas as pd\n\n# These library are for working with directories\nimport os\nfrom glob import glob\nfrom tqdm import tqdm\n\n# These library are for Visualization\nimport matplotlib.pyplot as plt\nimport plotly.express as px\n\n# These Library are for converting Label Encoding\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import OneHotEncoder\n\n# These library are for building model \nfrom tensorflow.keras import layers\nimport tensorflow.keras.backend as K\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.applications.imagenet_utils import preprocess_input\nfrom tensorflow.keras.layers import AveragePooling2D, MaxPooling2D, Dropout\nfrom tensorflow.keras.layers import Input, Dense, Activation, BatchNormalization, Flatten, Conv2D\nfrom tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping, ReduceLROnPlateau\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=DeprecationWarning)","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:56.849837Z","iopub.execute_input":"2022-02-25T09:32:56.850271Z","iopub.status.idle":"2022-02-25T09:32:56.859610Z","shell.execute_reply.started":"2022-02-25T09:32:56.850220Z","shell.execute_reply":"2022-02-25T09:32:56.858775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Getting the Files present in the main Directories\n\npath = '/kaggle/input/happy-whale-and-dolphin/'\nos.listdir(path)","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:56.861610Z","iopub.execute_input":"2022-02-25T09:32:56.862092Z","iopub.status.idle":"2022-02-25T09:32:56.873345Z","shell.execute_reply.started":"2022-02-25T09:32:56.862037Z","shell.execute_reply":"2022-02-25T09:32:56.872342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading the Train csv file and Sample Submission File using main dir\n\ntrain_data = pd.read_csv(path+'train.csv')\nsamp_subm = pd.read_csv(path+'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:56.874655Z","iopub.execute_input":"2022-02-25T09:32:56.875376Z","iopub.status.idle":"2022-02-25T09:32:56.958060Z","shell.execute_reply.started":"2022-02-25T09:32:56.875325Z","shell.execute_reply":"2022-02-25T09:32:56.957345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the dimension of the train.csv file\n\nprint('Number train samples:', len(train_data))","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:56.960417Z","iopub.execute_input":"2022-02-25T09:32:56.960820Z","iopub.status.idle":"2022-02-25T09:32:56.965972Z","shell.execute_reply.started":"2022-02-25T09:32:56.960783Z","shell.execute_reply":"2022-02-25T09:32:56.965216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying the column name present in the train.csv file\n\ntrain_data.columns","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:56.968479Z","iopub.execute_input":"2022-02-25T09:32:56.968998Z","iopub.status.idle":"2022-02-25T09:32:56.976123Z","shell.execute_reply.started":"2022-02-25T09:32:56.968884Z","shell.execute_reply":"2022-02-25T09:32:56.975243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying the first five in rows in the train.csv file\n\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:56.977700Z","iopub.execute_input":"2022-02-25T09:32:56.978154Z","iopub.status.idle":"2022-02-25T09:32:56.989337Z","shell.execute_reply.started":"2022-02-25T09:32:56.978110Z","shell.execute_reply":"2022-02-25T09:32:56.988423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the Number of training images available in the train image directory\n\nprint('Number train images:', len(os.listdir(path+'train_images/')))","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:56.991069Z","iopub.execute_input":"2022-02-25T09:32:56.991436Z","iopub.status.idle":"2022-02-25T09:32:57.021264Z","shell.execute_reply.started":"2022-02-25T09:32:56.991400Z","shell.execute_reply":"2022-02-25T09:32:57.020539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the Number of testing images available in the test image directory\n\nprint('Number test images:', len(os.listdir(path+'test_images/')))","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:57.022390Z","iopub.execute_input":"2022-02-25T09:32:57.022697Z","iopub.status.idle":"2022-02-25T09:32:57.039226Z","shell.execute_reply.started":"2022-02-25T09:32:57.022663Z","shell.execute_reply":"2022-02-25T09:32:57.038396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dispalying the different species availabel in the dataset\n\ntrain_data['species'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:57.040730Z","iopub.execute_input":"2022-02-25T09:32:57.040993Z","iopub.status.idle":"2022-02-25T09:32:57.055590Z","shell.execute_reply.started":"2022-02-25T09:32:57.040961Z","shell.execute_reply":"2022-02-25T09:32:57.054902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting using Pie Chart of the Species available in the dataset uisng Ploty\n\nfig = px.pie(train_data, values=train_data['species'].value_counts().values, names=train_data['species'].value_counts().index)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:57.058090Z","iopub.execute_input":"2022-02-25T09:32:57.058288Z","iopub.status.idle":"2022-02-25T09:32:57.116929Z","shell.execute_reply.started":"2022-02-25T09:32:57.058265Z","shell.execute_reply":"2022-02-25T09:32:57.116208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting BarChart of the Species avialable\n\nplt.figure(figsize=(15, 12))\nplt.rcParams[\"font.size\"] = 18\nplt.barh(train_data[\"species\"].value_counts().sort_values(ascending=True).index,\n         train_data[\"species\"].value_counts().sort_values(ascending=True),\n         tick_label = train_data[\"species\"].value_counts().sort_values(ascending=True).index)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:57.118127Z","iopub.execute_input":"2022-02-25T09:32:57.118553Z","iopub.status.idle":"2022-02-25T09:32:57.502760Z","shell.execute_reply.started":"2022-02-25T09:32:57.118518Z","shell.execute_reply":"2022-02-25T09:32:57.502050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying the individual id present in the dataset\n\ntrain_data['individual_id'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:57.505705Z","iopub.execute_input":"2022-02-25T09:32:57.505903Z","iopub.status.idle":"2022-02-25T09:32:57.526845Z","shell.execute_reply.started":"2022-02-25T09:32:57.505878Z","shell.execute_reply":"2022-02-25T09:32:57.526001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we will prepare our data for training and also plot few images \n\nBASE_PATH = \"../input/happy-whale-and-dolphin/train_images/\"\nTRAIN_IMAGES = glob(BASE_PATH + \"train/*.jpg\")","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:57.527869Z","iopub.execute_input":"2022-02-25T09:32:57.528485Z","iopub.status.idle":"2022-02-25T09:32:57.532631Z","shell.execute_reply.started":"2022-02-25T09:32:57.528446Z","shell.execute_reply":"2022-02-25T09:32:57.531822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying Images Randomly for training dataset\n\npath = BASE_PATH + np.random.choice(train_data['image'])\nim = plt.imread(path)\nplt.figure(figsize=(15, 6))\nplt.imshow(im)\nplt.title(path.split(\"/\")[-1])\nplt.xticks([]), plt.yticks([])\ntrain_data[train_data['image']==path.split('/')[-1]]\n","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:57.534163Z","iopub.execute_input":"2022-02-25T09:32:57.534743Z","iopub.status.idle":"2022-02-25T09:32:59.074809Z","shell.execute_reply.started":"2022-02-25T09:32:57.534703Z","shell.execute_reply":"2022-02-25T09:32:59.070433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Storing the Base path and then creating test images for further use\n\nBASE_PATH = \"../input/happy-whale-and-dolphin/test_images/\"\nTEST_IMAGES = glob(BASE_PATH + \"*.jpg\")","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:59.076104Z","iopub.execute_input":"2022-02-25T09:32:59.076762Z","iopub.status.idle":"2022-02-25T09:32:59.173581Z","shell.execute_reply.started":"2022-02-25T09:32:59.076725Z","shell.execute_reply":"2022-02-25T09:32:59.172950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying images randomly using test data\n\npath = np.random.choice(TEST_IMAGES)\nim = plt.imread(path)\nplt.figure(figsize=(15, 6))\nplt.imshow(im)\nplt.title(path.split(\"/\")[-1])\n","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:32:59.176040Z","iopub.execute_input":"2022-02-25T09:32:59.176244Z","iopub.status.idle":"2022-02-25T09:33:00.626695Z","shell.execute_reply.started":"2022-02-25T09:32:59.176220Z","shell.execute_reply":"2022-02-25T09:33:00.626095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating label in train.csv\n\ntrain_data['label'] = train_data.species.map(lambda x: 'whale' if 'whale' in x else 'dolphin')","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:33:00.627840Z","iopub.execute_input":"2022-02-25T09:33:00.628208Z","iopub.status.idle":"2022-02-25T09:33:00.650886Z","shell.execute_reply.started":"2022-02-25T09:33:00.628173Z","shell.execute_reply":"2022-02-25T09:33:00.650229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Barchart of Whale vs Dolphin count\n\ndata = train_data['label'].value_counts().reset_index()\nfig = px.bar(data, x='index', y='label', color='label', title='Whale Vs Dolphin', text_auto=True)\nfig.update_traces(textfont_size=12, textangle=0, textposition=\"outside\", cliponaxis=False)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:33:00.652009Z","iopub.execute_input":"2022-02-25T09:33:00.652343Z","iopub.status.idle":"2022-02-25T09:33:00.723302Z","shell.execute_reply.started":"2022-02-25T09:33:00.652305Z","shell.execute_reply":"2022-02-25T09:33:00.722673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting proportion of Whales vs Dolphins\n\nfig, ax  = plt.subplots(figsize=(16, 8))\nfig.suptitle('Whales and Dolphins ', size = 20, font=\"Serif\")\nexplode = (0.05, 0.05)\nlabels = list(train_data.label.value_counts().index)\nsizes = train_data.label.value_counts().values\nax.pie(sizes, explode=explode,startangle=60, labels=labels,autopct='%1.0f%%', pctdistance=0.7, colors=[\"#0077b6\",\"#90e0ef\"])\nax.add_artist(plt.Circle((0,0),0.4,fc='white'))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:33:00.724459Z","iopub.execute_input":"2022-02-25T09:33:00.724693Z","iopub.status.idle":"2022-02-25T09:33:00.857904Z","shell.execute_reply.started":"2022-02-25T09:33:00.724661Z","shell.execute_reply":"2022-02-25T09:33:00.857099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model building","metadata":{}},{"cell_type":"code","source":"#loading train.csv as train_df\n\ntrain_df = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:33:00.859192Z","iopub.execute_input":"2022-02-25T09:33:00.859449Z","iopub.status.idle":"2022-02-25T09:33:00.940101Z","shell.execute_reply.started":"2022-02-25T09:33:00.859415Z","shell.execute_reply":"2022-02-25T09:33:00.939403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the dimension of the train_df \n\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:33:00.941506Z","iopub.execute_input":"2022-02-25T09:33:00.941760Z","iopub.status.idle":"2022-02-25T09:33:00.948095Z","shell.execute_reply.started":"2022-02-25T09:33:00.941726Z","shell.execute_reply":"2022-02-25T09:33:00.947211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying First Five column of the train_df\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:33:00.949857Z","iopub.execute_input":"2022-02-25T09:33:00.950446Z","iopub.status.idle":"2022-02-25T09:33:00.962346Z","shell.execute_reply.started":"2022-02-25T09:33:00.950408Z","shell.execute_reply":"2022-02-25T09:33:00.961646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking for null values\n\ntrain_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:33:00.963635Z","iopub.execute_input":"2022-02-25T09:33:00.963875Z","iopub.status.idle":"2022-02-25T09:33:00.988461Z","shell.execute_reply.started":"2022-02-25T09:33:00.963826Z","shell.execute_reply":"2022-02-25T09:33:00.987738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Removing duplicate values form the train_df of column individual_id\n\ntrain_df=train_df.drop_duplicates(subset=['individual_id'],keep='last')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This function will Load Images of certain dimension\n\ndef Loading_Images(data, m, dataset):\n    print(\"Loading images\")\n    X_train = np.zeros((m, 32, 32, 3))\n    count = 0\n    for fig in tqdm(data['image']):\n        img = image.load_img(\"../input/happy-whale-and-dolphin/\"+dataset+\"/\"+fig, target_size=(32, 32, 3))\n        x = image.img_to_array(img)\n        x = preprocess_input(x)\n        X_train[count] = x\n        count += 1\n    return X_train","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:33:01.006967Z","iopub.execute_input":"2022-02-25T09:33:01.007158Z","iopub.status.idle":"2022-02-25T09:33:01.013827Z","shell.execute_reply.started":"2022-02-25T09:33:01.007129Z","shell.execute_reply":"2022-02-25T09:33:01.013137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This function will convert the text category to numeric\n\ndef prepare_labels(y):\n    values = np.array(y)\n    label_encoder = LabelEncoder()\n    integer_encoded = label_encoder.fit_transform(values)\n    onehot_encoder = OneHotEncoder(sparse=False)\n    integer_encoded = integer_encoded.reshape(len(integer_encoded), 1)\n    onehot_encoded = onehot_encoder.fit_transform(integer_encoded)\n    y = onehot_encoded\n    return y, label_encoder","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:33:01.015097Z","iopub.execute_input":"2022-02-25T09:33:01.015879Z","iopub.status.idle":"2022-02-25T09:33:01.024290Z","shell.execute_reply.started":"2022-02-25T09:33:01.015841Z","shell.execute_reply":"2022-02-25T09:33:01.023422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = Loading_Images(train_df, train_df.shape[0], \"train_images\")\nX /= 255","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:33:01.028632Z","iopub.execute_input":"2022-02-25T09:33:01.029153Z","iopub.status.idle":"2022-02-25T09:48:02.459549Z","shell.execute_reply.started":"2022-02-25T09:33:01.029112Z","shell.execute_reply":"2022-02-25T09:48:02.458811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y, label_encoder = prepare_labels(train_df['individual_id'])","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:48:02.460898Z","iopub.execute_input":"2022-02-25T09:48:02.461311Z","iopub.status.idle":"2022-02-25T09:48:02.559858Z","shell.execute_reply.started":"2022-02-25T09:48:02.461272Z","shell.execute_reply":"2022-02-25T09:48:02.559079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.shape","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:48:02.561314Z","iopub.execute_input":"2022-02-25T09:48:02.561604Z","iopub.status.idle":"2022-02-25T09:48:02.567913Z","shell.execute_reply.started":"2022-02-25T09:48:02.561568Z","shell.execute_reply":"2022-02-25T09:48:02.567077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Now that we have our data preprocessed we will build our model using Keras CNN","metadata":{}},{"cell_type":"code","source":"# Creating Model\n\nmodel = Sequential()\n\nmodel.add(Conv2D(32, (6, 6), strides = (1, 1), input_shape = (32, 32, 3)))\nmodel.add(BatchNormalization(axis = 3))\nmodel.add(Activation('relu'))\n\nmodel.add(MaxPooling2D((2, 2)))\nmodel.add(Conv2D(64, (3, 3), strides = (1,1)))\nmodel.add(Activation('relu'))\nmodel.add(AveragePooling2D((3, 3)))\n\nmodel.add(Flatten())\nmodel.add(Dense(512, activation=\"relu\"))\nmodel.add(Dropout(0.85))\n\nmodel.add(Dense(y.shape[1], activation='softmax'))\n\nmodel.compile(loss='categorical_crossentropy', optimizer=\"adam\", metrics=['accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:48:02.569226Z","iopub.execute_input":"2022-02-25T09:48:02.569500Z","iopub.status.idle":"2022-02-25T09:48:05.169930Z","shell.execute_reply.started":"2022-02-25T09:48:02.569465Z","shell.execute_reply":"2022-02-25T09:48:05.169206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model fitting \n\nhistory = model.fit(X, y, epochs=200, batch_size=128, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:48:05.171276Z","iopub.execute_input":"2022-02-25T09:48:05.171712Z","iopub.status.idle":"2022-02-25T09:54:31.688391Z","shell.execute_reply.started":"2022-02-25T09:48:05.171675Z","shell.execute_reply":"2022-02-25T09:54:31.687616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# saving our model for later use\n\nmodel.save('model.h5')","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:54:31.689855Z","iopub.execute_input":"2022-02-25T09:54:31.690186Z","iopub.status.idle":"2022-02-25T09:54:31.968308Z","shell.execute_reply.started":"2022-02-25T09:54:31.690144Z","shell.execute_reply":"2022-02-25T09:54:31.967488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation of the model ","metadata":{}},{"cell_type":"code","source":"# Plotting the accuracy of the model over the epochs\n\nplt.figure(figsize=(15,5))\nplt.plot(history.history['accuracy'])\nplt.title('Model accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:54:31.969757Z","iopub.execute_input":"2022-02-25T09:54:31.970022Z","iopub.status.idle":"2022-02-25T09:54:32.175855Z","shell.execute_reply.started":"2022-02-25T09:54:31.969968Z","shell.execute_reply":"2022-02-25T09:54:32.175142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting the loss of the model over the epochs\n\nplt.figure(figsize=(15,5))\nplt.plot(history.history['loss'])\nplt.title('Model loss')\nplt.ylabel('loss')\nplt.xlabel('Epoch')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:54:32.177079Z","iopub.execute_input":"2022-02-25T09:54:32.177438Z","iopub.status.idle":"2022-02-25T09:54:32.376207Z","shell.execute_reply.started":"2022-02-25T09:54:32.177402Z","shell.execute_reply":"2022-02-25T09:54:32.375498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"test = os.listdir(\"../input/happy-whale-and-dolphin/test_images\")\nprint(len(test))","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:54:32.377299Z","iopub.execute_input":"2022-02-25T09:54:32.378102Z","iopub.status.idle":"2022-02-25T09:54:32.399097Z","shell.execute_reply.started":"2022-02-25T09:54:32.378064Z","shell.execute_reply":"2022-02-25T09:54:32.398417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = ['image']\ntest_df = pd.DataFrame(test, columns=col)\ntest_df['predictions'] = ''\n#test_df=test_df.head(n=250)","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:54:32.400484Z","iopub.execute_input":"2022-02-25T09:54:32.400750Z","iopub.status.idle":"2022-02-25T09:54:32.407622Z","shell.execute_reply.started":"2022-02-25T09:54:32.400715Z","shell.execute_reply":"2022-02-25T09:54:32.406796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size=5000\nbatch_start = 0\nbatch_end = batch_size\nL = len(test_df)\n\nwhile batch_start < L:\n    limit = min(batch_end, L)\n    test_df_batch = test_df.iloc[batch_start:limit]\n    print(type(test_df_batch))\n    X = Loading_Images(test_df_batch, test_df_batch.shape[0], \"test_images\")\n    X /= 255\n    predictions = model.predict(np.array(X), verbose=1)\n    for i, pred in enumerate(predictions):\n        p=pred.argsort()[-5:][::-1]\n        idx=-1\n        s=''\n        s1=''\n        s2=''\n        for x in p:\n            idx=idx+1\n            if pred[x]>0.6:\n                s1 = s1 + ' ' +  label_encoder.inverse_transform(p)[idx]\n            else:\n                s2 = s2 + ' ' + label_encoder.inverse_transform(p)[idx]\n        s= s1 + ' new_individual' + s2\n        s = s.strip(' ')\n        test_df.loc[ batch_start + i, 'predictions'] = s\n    batch_start += batch_size   \n    batch_end += batch_size\n    ","metadata":{"execution":{"iopub.status.busy":"2022-02-25T09:54:32.408742Z","iopub.execute_input":"2022-02-25T09:54:32.409141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- For Submission we will create submission.csv","metadata":{}},{"cell_type":"code","source":"# Creating submission.csv and printing first five rows\n\ntest_df.to_csv('submission.csv',index=False)\ntest_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> That's it,<br>\n> I will continue to update the notebook,<br>\n> I know we can update so many things to improve overall accuracy and submission score,<br>\n> Let me know your suggestion :)","metadata":{}},{"cell_type":"markdown","source":"![](https://thumbs.dreamstime.com/b/dental-smile-whale-icon-cartoon-style-dental-smile-whale-icon-cartoon-dental-smile-whale-vector-icon-web-design-117675439.jpg)","metadata":{}}]}