{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Load important libraries\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom keras.models import Sequential\nfrom keras.preprocessing.image import ImageDataGenerator\nimport keras.layers as L\nfrom keras import regularizers, optimizers\nfrom collections import Counter\nimport keras\nfrom keras import Model\nimport tensorflow as tf\nfrom tensorflow.keras.applications.xception import Xception\nfrom keras.callbacks import EarlyStopping, ModelCheckpoint\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom keras.models import load_model","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T03:45:20.267357Z","iopub.execute_input":"2022-08-11T03:45:20.267742Z","iopub.status.idle":"2022-08-11T03:45:28.877269Z","shell.execute_reply.started":"2022-08-11T03:45:20.267663Z","shell.execute_reply":"2022-08-11T03:45:28.875823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the training data \nlabel = pd.read_csv(\"../input/landmark-recognition-2020/train.csv\")\nlabel.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:28.880632Z","iopub.execute_input":"2022-08-11T03:45:28.881453Z","iopub.status.idle":"2022-08-11T03:45:30.413334Z","shell.execute_reply.started":"2022-08-11T03:45:28.881405Z","shell.execute_reply":"2022-08-11T03:45:30.412019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Print the total number of pictures and landmarks\nprint(\"The total number of pictures in the dataset:\", len(label))\nprint(\"The total number of landmarks in the dataset:\", label.landmark_id.nunique())","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:30.415223Z","iopub.execute_input":"2022-08-11T03:45:30.416001Z","iopub.status.idle":"2022-08-11T03:45:30.458022Z","shell.execute_reply.started":"2022-08-11T03:45:30.415921Z","shell.execute_reply":"2022-08-11T03:45:30.456505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# missing data in training data \ntotal = label.isnull().sum().sort_values(ascending = False)\npercent = (label.isnull().sum()/label.isnull().count()).sort_values(ascending = False)\nmissing_train_data = pd.concat([total, percent], axis=1, keys=['Total', 'Percent'])\nmissing_train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:30.461840Z","iopub.execute_input":"2022-08-11T03:45:30.462766Z","iopub.status.idle":"2022-08-11T03:45:30.706427Z","shell.execute_reply.started":"2022-08-11T03:45:30.462721Z","shell.execute_reply":"2022-08-11T03:45:30.705012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So, we do not having any missing data.\n\nLet's move to find the top 10 landmarks with highest number of images:","metadata":{}},{"cell_type":"code","source":"# Top 10 landmark_id with highest number of landsmark images\ntop10 = pd.DataFrame(label.landmark_id.value_counts().head(10))\ntop10.reset_index(inplace=True)\ntop10.columns = ['landmark_id','count']\ntop10","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:30.708394Z","iopub.execute_input":"2022-08-11T03:45:30.709875Z","iopub.status.idle":"2022-08-11T03:45:30.769425Z","shell.execute_reply.started":"2022-08-11T03:45:30.709816Z","shell.execute_reply":"2022-08-11T03:45:30.768124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"percent=pd.DataFrame(((label.landmark_id.value_counts()/label.landmark_id.count())*100).head(10))\npercent.reset_index(inplace=True)\npercent.columns = ['landmark_id','percent_top10']\npercent","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:30.771600Z","iopub.execute_input":"2022-08-11T03:45:30.772135Z","iopub.status.idle":"2022-08-11T03:45:30.838642Z","shell.execute_reply.started":"2022-08-11T03:45:30.772043Z","shell.execute_reply":"2022-08-11T03:45:30.837028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the count and percentage, we can see that landmark_id: 138982, 126637, and 20409 are the three top landmark ids with highest number of images present in the training data.","metadata":{}},{"cell_type":"code","source":"# Plot the most frequent landmark_ids\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nplt.figure(figsize = (9, 8))\nplt.title('Most Frequent Landmarks')\nsns.set_color_codes(\"muted\")\nsns.barplot(x=\"landmark_id\", y=\"count\", data=top10, label=\"Count\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:30.840906Z","iopub.execute_input":"2022-08-11T03:45:30.841450Z","iopub.status.idle":"2022-08-11T03:45:31.715459Z","shell.execute_reply.started":"2022-08-11T03:45:30.841408Z","shell.execute_reply":"2022-08-11T03:45:31.714024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, let's analyze the bottom10 landmark_id with less number of landsmark images.","metadata":{}},{"cell_type":"code","source":"# Bottom 10 landmark_id with less number of landsmark images\nbot10 = pd.DataFrame(label.landmark_id.value_counts().tail(10))\nbot10.reset_index(inplace=True)\nbot10.columns = ['landmark_id','count']\nbot10","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:31.717478Z","iopub.execute_input":"2022-08-11T03:45:31.717825Z","iopub.status.idle":"2022-08-11T03:45:31.782746Z","shell.execute_reply.started":"2022-08-11T03:45:31.717795Z","shell.execute_reply":"2022-08-11T03:45:31.781160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"percent=pd.DataFrame(((label.landmark_id.value_counts()/label.landmark_id.count())*100).tail(10))\npercent.reset_index(inplace=True)\npercent.columns = ['landmark_id','percent_bottom10']\npercent","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:31.785032Z","iopub.execute_input":"2022-08-11T03:45:31.785532Z","iopub.status.idle":"2022-08-11T03:45:31.849289Z","shell.execute_reply.started":"2022-08-11T03:45:31.785488Z","shell.execute_reply":"2022-08-11T03:45:31.848011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So, we can see that there are many landmarks with only 2 images in the dataset.","metadata":{}},{"cell_type":"code","source":"# Plot the least frequent landmark_ids\nplt.figure(figsize = (9, 8))\nplt.title('Least Frequent Landmarks')\nsns.set_color_codes(\"muted\")\nsns.barplot(x=\"landmark_id\", y=\"count\", data=bot10, label=\"Count\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:31.855230Z","iopub.execute_input":"2022-08-11T03:45:31.855560Z","iopub.status.idle":"2022-08-11T03:45:32.155437Z","shell.execute_reply.started":"2022-08-11T03:45:31.855532Z","shell.execute_reply":"2022-08-11T03:45:32.154014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's analyse the number of images per landmark:","metadata":{}},{"cell_type":"code","source":"sns.set()\nplt.title('Training set: number of images per class(line plot)')\nlandmarks_fold = pd.DataFrame(label['landmark_id'].value_counts())\nlandmarks_fold.reset_index(inplace=True)\nlandmarks_fold.columns = ['landmark_id','count']\nax = landmarks_fold['count'].plot(logy=True, grid=True)\nlocs, labels = plt.xticks()\nplt.setp(labels, rotation=30)\nax.set(xlabel=\"Landmarks\", ylabel=\"Number of images\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:32.157345Z","iopub.execute_input":"2022-08-11T03:45:32.158417Z","iopub.status.idle":"2022-08-11T03:45:32.982457Z","shell.execute_reply.started":"2022-08-11T03:45:32.158373Z","shell.execute_reply":"2022-08-11T03:45:32.981129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So, we can see that there are so many landmarks that have very less images and if we include these landmarks, our model would not be able to predict them. Also, we want to work on a subset of data, so we are taking a cut-off of 400, which means that we are only sleecting landmarks with more than 400 images.","metadata":{}},{"cell_type":"code","source":"#Number of landmarks with less than or equal to 400 images:\ncounts = label['landmark_id'].value_counts().sort_values(ascending=False)\nbelow = counts[counts <= 400].index.shape[0]\nbelow","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:32.984348Z","iopub.execute_input":"2022-08-11T03:45:32.985151Z","iopub.status.idle":"2022-08-11T03:45:33.044996Z","shell.execute_reply.started":"2022-08-11T03:45:32.985082Z","shell.execute_reply":"2022-08-11T03:45:33.043230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We have 81233 classes with less than or equal to 400 images. Including these images in the model may create noise\n# as the number of these images are very less. therefore removing these images from our original dataset\n\nselected_classes = counts[counts > 400].index\nselected_df = label.loc[label.landmark_id.isin(selected_classes)]\nprint(\"The total number of pictures in the dataset:\", len(selected_df))\nprint(\"The total number of landmarks in the dataset:\", selected_df.landmark_id.nunique())","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:33.049060Z","iopub.execute_input":"2022-08-11T03:45:33.049424Z","iopub.status.idle":"2022-08-11T03:45:33.085548Z","shell.execute_reply.started":"2022-08-11T03:45:33.049394Z","shell.execute_reply":"2022-08-11T03:45:33.084215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we have 80 landmarks and 58479 observations/images.\n\nTo make the data balances, we want each landmark to have same number of images. So, we are taking 200 images for each landmark.","metadata":{}},{"cell_type":"code","source":"selected_df=selected_df.groupby('landmark_id').sample(200, random_state=42, replace=True)\nprint(\"The total number of pictures in the dataset:\", len(selected_df))\nprint(\"The total number of landmarks in the dataset:\", selected_df.landmark_id.nunique())","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:33.087823Z","iopub.execute_input":"2022-08-11T03:45:33.088345Z","iopub.status.idle":"2022-08-11T03:45:33.145916Z","shell.execute_reply.started":"2022-08-11T03:45:33.088272Z","shell.execute_reply":"2022-08-11T03:45:33.144497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our final dataset has 80 landmarks and 16000 images.","metadata":{}},{"cell_type":"code","source":"selected_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:33.147926Z","iopub.execute_input":"2022-08-11T03:45:33.148421Z","iopub.status.idle":"2022-08-11T03:45:33.170699Z","shell.execute_reply.started":"2022-08-11T03:45:33.148379Z","shell.execute_reply":"2022-08-11T03:45:33.168410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we are adding path to landmark images:","metadata":{}},{"cell_type":"code","source":"#Adding path of landmark images \ndef get_train_file_path(image_id):\n    return \"../input/landmark-recognition-2020/train/{}/{}/{}/{}.jpg\".format(image_id[0], image_id[1], image_id[2], image_id)\nselected_df['file_path'] = selected_df['id'].apply(get_train_file_path)\nselected_df=selected_df.reset_index()\nselected_df.drop(\"index\",axis=1,inplace=True)\nselected_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:33.172578Z","iopub.execute_input":"2022-08-11T03:45:33.173183Z","iopub.status.idle":"2022-08-11T03:45:33.209713Z","shell.execute_reply.started":"2022-08-11T03:45:33.173137Z","shell.execute_reply":"2022-08-11T03:45:33.207772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The below is the visualization to show that our data is balanced and each landmark has same number of images:","metadata":{}},{"cell_type":"code","source":"landmarks_fold = pd.DataFrame(selected_df['landmark_id'].value_counts())\nlandmarks_fold.reset_index(inplace=True)\nlandmarks_fold.columns = ['landmark_id','count']\n# Plot the most frequent landmark_ids\ndata10=landmarks_fold.head(5)\nplt.figure(figsize = (9, 8))\nplt.title('Most Frequent Landmarks')\nsns.set_color_codes(\"muted\")\nsns.barplot(x=\"landmark_id\", y=\"count\", data=data10, label=\"Count\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:33.211346Z","iopub.execute_input":"2022-08-11T03:45:33.212219Z","iopub.status.idle":"2022-08-11T03:45:33.498492Z","shell.execute_reply.started":"2022-08-11T03:45:33.212175Z","shell.execute_reply":"2022-08-11T03:45:33.497043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As remember, from our original data, we have landmark_id:138982,126637, and 20409 with highest number of pictures. So let's explore them:","metadata":{}},{"cell_type":"code","source":"#Lets examine landsmark is 138982 which is having highest count in our selected data:\nimport os\nimport glob\nimport cv2\ntrain = selected_df[selected_df.landmark_id==138982]\nfig = plt.figure(figsize=(30,20))\nx=1\nfor i in train.file_path[:16]:\n    image = cv2.imread(i)\n    fig.add_subplot(4, 4, x)\n    plt.imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n    plt.axis('off')\n    x+=1","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:33.500791Z","iopub.execute_input":"2022-08-11T03:45:33.501269Z","iopub.status.idle":"2022-08-11T03:45:37.723498Z","shell.execute_reply.started":"2022-08-11T03:45:33.501194Z","shell.execute_reply":"2022-08-11T03:45:37.721319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets examine landsmark is 126637 which is having highest count in our selected data:\nimport os\nimport glob\nimport cv2\ntrain = selected_df[selected_df.landmark_id==126637]\nfig = plt.figure(figsize=(30,20))\nx=1\nfor i in train.file_path[:16]:\n    image = cv2.imread(i)\n    fig.add_subplot(4, 4, x)\n    plt.imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n    plt.axis('off')\n    x+=1","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:37.725350Z","iopub.execute_input":"2022-08-11T03:45:37.726149Z","iopub.status.idle":"2022-08-11T03:45:41.576410Z","shell.execute_reply.started":"2022-08-11T03:45:37.726062Z","shell.execute_reply":"2022-08-11T03:45:41.574878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n#Lets examine landsmark is 20409 which is having highest count in our selected data:\nimport os\nimport glob\nimport cv2\ntrain = selected_df[selected_df.landmark_id==20409]\nfig = plt.figure(figsize=(30,20))\nx=1\nfor i in train.file_path[:16]:\n    image = cv2.imread(i)\n    fig.add_subplot(4, 4, x)\n    plt.imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n    plt.axis('off')\n    x+=1","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:41.578133Z","iopub.execute_input":"2022-08-11T03:45:41.580285Z","iopub.status.idle":"2022-08-11T03:45:44.896132Z","shell.execute_reply.started":"2022-08-11T03:45:41.580242Z","shell.execute_reply":"2022-08-11T03:45:44.893043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Converting landmark_id to string for model:\nselected_df['landmark_id'] = selected_df.landmark_id.astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:44.898378Z","iopub.execute_input":"2022-08-11T03:45:44.899142Z","iopub.status.idle":"2022-08-11T03:45:44.933530Z","shell.execute_reply.started":"2022-08-11T03:45:44.899089Z","shell.execute_reply":"2022-08-11T03:45:44.932272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:44.935531Z","iopub.execute_input":"2022-08-11T03:45:44.936343Z","iopub.status.idle":"2022-08-11T03:45:44.962398Z","shell.execute_reply.started":"2022-08-11T03:45:44.936278Z","shell.execute_reply":"2022-08-11T03:45:44.961247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We are splitting the dataset into train and test with 80:20 ratio and startify as landmark_id, so that both test and train ahs equal number of landmark classes:","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain,test=train_test_split(selected_df,test_size=0.2, stratify=selected_df[\"landmark_id\"])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:44.964289Z","iopub.execute_input":"2022-08-11T03:45:44.965179Z","iopub.status.idle":"2022-08-11T03:45:45.132794Z","shell.execute_reply.started":"2022-08-11T03:45:44.965119Z","shell.execute_reply":"2022-08-11T03:45:45.131409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:45.135101Z","iopub.execute_input":"2022-08-11T03:45:45.135627Z","iopub.status.idle":"2022-08-11T03:45:45.153177Z","shell.execute_reply.started":"2022-08-11T03:45:45.135580Z","shell.execute_reply":"2022-08-11T03:45:45.151389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"landmark_id\"].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:45.155077Z","iopub.execute_input":"2022-08-11T03:45:45.156760Z","iopub.status.idle":"2022-08-11T03:45:45.173512Z","shell.execute_reply.started":"2022-08-11T03:45:45.156714Z","shell.execute_reply":"2022-08-11T03:45:45.171846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:45.176222Z","iopub.execute_input":"2022-08-11T03:45:45.176793Z","iopub.status.idle":"2022-08-11T03:45:45.195230Z","shell.execute_reply.started":"2022-08-11T03:45:45.176749Z","shell.execute_reply":"2022-08-11T03:45:45.193633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"landmark_id\"].nunique()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:45.197776Z","iopub.execute_input":"2022-08-11T03:45:45.198519Z","iopub.status.idle":"2022-08-11T03:45:45.211742Z","shell.execute_reply.started":"2022-08-11T03:45:45.198476Z","shell.execute_reply":"2022-08-11T03:45:45.208141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we are splitting the training dataset into train and validation with validation rate 0.2:","metadata":{}},{"cell_type":"code","source":"val_rate = 0.2\nbatch_size = 32\ngen = ImageDataGenerator(rescale=1./255,validation_split=val_rate)\n\ntrain_gen = gen.flow_from_dataframe(\n    train,\n    x_col=\"file_path\",\n    y_col=\"landmark_id\",\n    weight_col=None,\n    target_size=(256, 256),\n    color_mode=\"rgb\",\n    classes=None,\n    class_mode=\"categorical\",\n    batch_size=batch_size,\n    shuffle=True,\n    subset=\"training\",\n    seed=123,\n    interpolation=\"nearest\",\n    validate_filenames=False)\n    \nval_gen = gen.flow_from_dataframe(\n    train,\n    x_col=\"file_path\",\n    y_col=\"landmark_id\",\n    weight_col=None,\n    target_size=(256, 256),\n    color_mode=\"rgb\",\n    classes=None,\n    class_mode=\"categorical\",\n    batch_size=batch_size,\n    shuffle=True,\n    subset=\"validation\",\n    seed=123,\n    interpolation=\"nearest\",\n    validate_filenames=False)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:45.221400Z","iopub.execute_input":"2022-08-11T03:45:45.221743Z","iopub.status.idle":"2022-08-11T03:45:45.301475Z","shell.execute_reply.started":"2022-08-11T03:45:45.221713Z","shell.execute_reply":"2022-08-11T03:45:45.300142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Making test dataset also in same format:\ngen = ImageDataGenerator(rescale=1./255)\ntest_gen = gen.flow_from_dataframe(\n    test,\n    x_col=\"file_path\",\n    y_col=\"landmark_id\",\n    target_size=(256, 256),\n    batch_size =1,\n    class_mode=\"categorical\",\n    seed=123)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:45.303257Z","iopub.execute_input":"2022-08-11T03:45:45.303612Z","iopub.status.idle":"2022-08-11T03:45:54.404898Z","shell.execute_reply.started":"2022-08-11T03:45:45.303582Z","shell.execute_reply":"2022-08-11T03:45:54.403518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Importing ResNet50 and including paarmeter\nfrom tensorflow.keras.applications import ResNet50V2\nconv_base = ResNet50V2(include_top=False,\n                    weights=\"imagenet\",\n                    input_shape=(256, 256, 3))\nconv_base.trainable = True\nconv_base.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:45:54.406866Z","iopub.execute_input":"2022-08-11T03:45:54.407304Z","iopub.status.idle":"2022-08-11T03:46:02.822830Z","shell.execute_reply.started":"2022-08-11T03:45:54.407273Z","shell.execute_reply":"2022-08-11T03:46:02.819262Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Making model\nfrom tensorflow.keras.layers import Dense, Dropout, MaxPooling2D, GlobalAveragePooling2D, Flatten, Conv2D, Input\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras import optimizers\nimport tensorflow as tf\n\nmodel = Sequential()\nmodel.add(conv_base)\nmodel.add(GlobalAveragePooling2D())\nmodel.add(Dense(80, activation='sigmoid'))\nmodel.compile(optimizer='adamax',\n              loss = 'categorical_crossentropy',\n              metrics=[tf.keras.metrics.Precision(),tf.keras.metrics.Recall(),'accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:46:02.825198Z","iopub.execute_input":"2022-08-11T03:46:02.825679Z","iopub.status.idle":"2022-08-11T03:46:03.345664Z","shell.execute_reply.started":"2022-08-11T03:46:02.825620Z","shell.execute_reply":"2022-08-11T03:46:03.344049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Case1: Taking epochs=5**","metadata":{}},{"cell_type":"code","source":"#setting epochs, train_steps, and val_steps\nepochs = 5\ntrain_steps = int(len(train)*(1-val_rate))//batch_size\nval_steps = int(len(train)*val_rate)//batch_size","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:46:03.349630Z","iopub.execute_input":"2022-08-11T03:46:03.349944Z","iopub.status.idle":"2022-08-11T03:46:03.358691Z","shell.execute_reply.started":"2022-08-11T03:46:03.349915Z","shell.execute_reply":"2022-08-11T03:46:03.356811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting the model\nhistory = model.fit_generator(train_gen, \n                              steps_per_epoch=train_steps, \n                              epochs=epochs,validation_data=val_gen, \n                              validation_steps=val_steps)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:48:28.503865Z","iopub.execute_input":"2022-08-11T03:48:28.504348Z","iopub.status.idle":"2022-08-11T04:02:07.527492Z","shell.execute_reply.started":"2022-08-11T03:48:28.504315Z","shell.execute_reply":"2022-08-11T04:02:07.525795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plotting Accuracy and Precision for training and validation\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nprecision = history.history['precision']\nval_precision = history.history['val_precision']\n\nepochs = range(1, len(acc)+1)\n\nplt.plot(epochs, acc, '#21466C', label='Training acc')\nplt.plot(epochs, val_acc, '#cc1123', label='Validation acc')\nplt.xlabel('num of Epochs')\nplt.ylabel('accuracy')\nplt.title('Training and validation accuracy')\nplt.legend()\n\nplt.figure()\n\nplt.plot(epochs, precision, '#21466C', label='Training precision')\nplt.plot(epochs, val_precision, '#cc1123', label='Validation precision')\nplt.xlabel('num of Epochs')\nplt.ylabel('precision')\nplt.title('Training and validation precision')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:03:13.744273Z","iopub.execute_input":"2022-08-11T04:03:13.744772Z","iopub.status.idle":"2022-08-11T04:03:14.439882Z","shell.execute_reply.started":"2022-08-11T04:03:13.744733Z","shell.execute_reply":"2022-08-11T04:03:14.438293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.evaluate(test_gen)\nprint('loss:', scores[0])\nprint('precision:', scores[1])\nprint('recall:', scores[2])\nprint('accuracy:', scores[3])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:03:28.620422Z","iopub.execute_input":"2022-08-11T04:03:28.620827Z","iopub.status.idle":"2022-08-11T04:04:50.619556Z","shell.execute_reply.started":"2022-08-11T04:03:28.620796Z","shell.execute_reply":"2022-08-11T04:04:50.617165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Case2: Taking Epochs=10**","metadata":{}},{"cell_type":"code","source":"#setting epochs, train_steps, and val_steps\nepochs = 10\ntrain_steps = int(len(train)*(1-val_rate))//batch_size\nval_steps = int(len(train)*val_rate)//batch_size\n#Fitting the model\nhistory = model.fit_generator(train_gen, \n                              steps_per_epoch=train_steps, \n                              epochs=epochs,validation_data=val_gen, \n                              validation_steps=val_steps)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:05:11.259945Z","iopub.execute_input":"2022-08-11T04:05:11.260416Z","iopub.status.idle":"2022-08-11T04:29:38.348075Z","shell.execute_reply.started":"2022-08-11T04:05:11.260383Z","shell.execute_reply":"2022-08-11T04:29:38.346535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plotting Accuracy and Precision for training and validation\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nprecision = history.history['precision']\nval_precision = history.history['val_precision']\n\nepochs = range(1, len(acc)+1)\n\nplt.plot(epochs, acc, '#21466C', label='Training acc')\nplt.plot(epochs, val_acc, '#cc1123', label='Validation acc')\nplt.xlabel('num of Epochs')\nplt.ylabel('accuracy')\nplt.title('Training and validation accuracy')\nplt.legend()\n\nplt.figure()\n\nplt.plot(epochs, precision, '#21466C', label='Training precision')\nplt.plot(epochs, val_precision, '#cc1123', label='Validation precision')\nplt.xlabel('num of Epochs')\nplt.ylabel('precision')\nplt.title('Training and validation precision')\nplt.legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:36:33.858145Z","iopub.execute_input":"2022-08-11T04:36:33.858622Z","iopub.status.idle":"2022-08-11T04:36:34.545545Z","shell.execute_reply.started":"2022-08-11T04:36:33.858580Z","shell.execute_reply":"2022-08-11T04:36:34.544024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores = model.evaluate(test_gen)\nprint('loss:', scores[0])\nprint('precision:', scores[1])\nprint('recall:', scores[2])\nprint('accuracy:', scores[3])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T04:36:41.250830Z","iopub.execute_input":"2022-08-11T04:36:41.251315Z","iopub.status.idle":"2022-08-11T04:38:03.241851Z","shell.execute_reply.started":"2022-08-11T04:36:41.251282Z","shell.execute_reply":"2022-08-11T04:38:03.240346Z"},"trusted":true},"execution_count":null,"outputs":[]}]}