{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os\nimport random\nimport seaborn as sns\nimport cv2\n\n# General packages\nimport pandas as pd\nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport PIL\nimport IPython.display as ipd\nimport glob\nimport h5py\nimport plotly.graph_objs as go\nimport plotly.express as px\nfrom PIL import Image\nfrom tempfile import mktemp\n\nfrom bokeh.layouts import column, row\nfrom bokeh.models import ColumnDataSource, LinearAxis, Range1d\nfrom bokeh.models.tools import HoverTool\nfrom bokeh.palettes import BuGn4\nfrom bokeh.plotting import figure, output_notebook, show\nfrom bokeh.transform import cumsum\nfrom math import pi\n\noutput_notebook()\n\nfrom IPython.display import Image, display\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom keras.models import load_model\nfrom keras.preprocessing import image\nfrom PIL import Image","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(os.listdir('../input/landmark-recognition-2020/'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"BASE_PATH = '../input/landmark-recognition-2020'\n\nTRAIN_DIR = f'{BASE_PATH}/train'\nTEST_DIR = f'{BASE_PATH}/test'\n\nprint('Reading data...')\ntrain = pd.read_csv(f'{BASE_PATH}/train.csv')\nsub = pd.read_csv(f'{BASE_PATH}/sample_submission.csv')\nprint('Reading data completed')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train.shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sub.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"landmarks = len(train['landmark_id'].unique())\nlandmarks","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print('Top few landmark_ids by count')\n\nz = train.landmark_id.value_counts().head(10).to_frame()\nz.reset_index(inplace=True)\nz.columns=['landmark_id','count']\nz.landmark_id = z.landmark_id.apply(lambda x: f'id_{x}')\n\nz.style.background_gradient(cmap='Oranges')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# displaying only top 30 landmark\nlandmark = train.landmark_id.value_counts()\nlandmark_df = pd.DataFrame({'landmark_id':landmark.index, 'frequency':landmark.values}).head(30)\n\nlandmark_df['landmark_id'] =   landmark_df.landmark_id.apply(lambda x: f'landmark_id_{x}')\n\nfig = px.bar(landmark_df, x=\"frequency\", y=\"landmark_id\",color='landmark_id', orientation='h',\n             hover_data=[\"landmark_id\", \"frequency\"],\n             height=1000,\n             title='Number of images per landmark_id (Top 30 landmark_ids)')\nfig.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import PIL\nfrom PIL import Image, ImageDraw\n\n\ndef display_images(images, title=None): \n    f, ax = plt.subplots(5,5, figsize=(18,22))\n    if title:\n        f.suptitle(title, fontsize = 30)\n\n    for i, image_id in enumerate(images):\n        image_path = os.path.join(TRAIN_DIR, f'{image_id[0]}/{image_id[1]}/{image_id[2]}/{image_id}.jpg')\n        image = Image.open(image_path)\n        \n        ax[i//5, i%5].imshow(image) \n        image.close()       \n        ax[i//5, i%5].axis('off')\n\n        landmark_id = train[train.id==image_id.split('.')[0]].landmark_id.values[0]\n        ax[i//5, i%5].set_title(f\"ID: {image_id.split('.')[0]}\\nLandmark_id: {landmark_id}\", fontsize=\"12\")\n\n    plt.show() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"samples = train.sample(25).id.values\ndisplay_images(samples)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"samples = train[train.landmark_id == 138982].sample(25).id.values\ndisplay_images(samples)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Most Occuring Landmarks","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"from collections import Counter\nlandmark_counts = dict(Counter(train['landmark_id']))\nlandmark_dict = {'landmark_id': list(landmark_counts.keys()), 'count': list(landmark_counts.values())}\n\nlandmark_count_df = pd.DataFrame.from_dict(landmark_dict)\nlandmark_count_sorted = landmark_count_df.sort_values('count', ascending = False)\nlandmark_count_sorted.head(20)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Distribution of Landmarks with their counts","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"fig_count = px.histogram(landmark_count_df, x = 'landmark_id', y = 'count')\nfig_count.update_layout(\n    title_text='Distribution of Landmarks',\n    xaxis_title_text='Landmark ID',\n    yaxis_title_text='Count'\n)\n\nfig_count.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Pre-trained models","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.applications import(\n                vgg16,\n                resnet50,\n                mobilenet,\n                inception_v3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"vgg_model = vgg16.VGG16(weights = 'imagenet')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"resnet_model = resnet50.ResNet50(weights = 'imagenet')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"mobilenet_model = mobilenet.MobileNet(weights = 'imagenet') ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_list = glob.glob('../input/landmark-recognition-2020/train/*/*/*/*')\ntest_list = glob.glob('../input/landmark-recognition-2020/test/*/*/*/*')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_list","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_list","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"filename = '../input/landmark-recognition-2020/train/1/1/1/11172998c813fe6f.jpg'","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# VGG16","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"original = image.load_img(filename,target_size=(224,224))\nprint('PIL image size',original.size)\nplt.imshow(original)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import img_to_array\nnumpy_image = img_to_array(original)\nplt.imshow(np.uint8(numpy_image))\nplt.show()\nprint('numpy array size',numpy_image.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_batch = np.expand_dims(numpy_image, axis=0)\nprint('image batch size', image_batch.shape)\nplt.imshow(np.uint8(image_batch[0]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# prepare the image for the VGG model\nfrom keras.applications.imagenet_utils import decode_predictions\nprocessed_image = vgg16.preprocess_input(image_batch.copy())\n# get the predicted probabilities for each class\npredictions = vgg_model.predict(processed_image)\n# print predictions\n# convert the probabilities to class labels\n# we will get top 5 predictions which is the default\nlabel_vgg = decode_predictions(predictions)\n# print VGG16 predictions\nfor prediction_id in range(len(label_vgg[0])):\n    print(label_vgg[0][prediction_id])\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# ResNet50","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"from keras.applications.resnet50 import preprocess_input\nfrom keras.applications.imagenet_utils import decode_predictions","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"img = image.load_img(filename,target_size=(224,224))\nimg = image.img_to_array(img)\nimg = np.expand_dims(img,axis=0)\nimg = preprocess_input(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = resnet_model.predict(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print( decode_predictions(preds, top=1)[0])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# MobileNet ","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"pred = mobilenet_model.predict(img)\nprint(decode_predictions(pred))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}