{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n#         print(os.path.join(dirname, filename))\n        continue\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-19T14:13:48.749494Z","iopub.execute_input":"2022-04-19T14:13:48.749811Z","iopub.status.idle":"2022-04-19T14:14:24.793644Z","shell.execute_reply.started":"2022-04-19T14:13:48.749724Z","shell.execute_reply":"2022-04-19T14:14:24.792805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Landmark Recognition**","metadata":{}},{"cell_type":"code","source":"import os\nimport random\nimport seaborn as sns\nimport cv2\n\nimport pandas as pd\nimport numpy as np\nimport matplotlib\nimport matplotlib.pyplot as plt\nimport PIL\nimport IPython.display as ipd\nimport glob\nimport h5py\nimport plotly.graph_objs as go\nimport plotly.express as px\nfrom PIL import Image\nfrom tempfile import mktemp\nfrom bokeh.plotting import figure, output_notebook, show\nfrom math import pi\n\nfrom tqdm import tqdm\nfrom tqdm.notebook import tqdm_notebook\ntqdm_notebook.pandas()\n\noutput_notebook()\n\nfrom IPython.display import Image, display","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_DIR = '../input/landmark-recognition-2021'\n\nTRAIN_IMAGE_DIR = f'{DATASET_DIR}/train'\nTEST_IMAGE_DIR = f'{DATASET_DIR}/test'\n\ntrain = pd.read_csv(f'{DATASET_DIR}/train.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.head())\nprint(\"Training data shape :\", train.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.isnull().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"value_counts = train['landmark_id'].value_counts() # normalize=True returns relative frequency\n\nfreq_df = pd.DataFrame(value_counts)\nfreq_df.reset_index(inplace=True)\nfreq_df.columns = ['landmark_id','frequency']\nfreq_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Prepare dataset\n\nThere is a total of 81313 different classes for landmarks. Because this is a great amount we were planning to only take an x percentage of each class and delete classes with less than x frequency. This plan wont work out because more than 41k classes have less than 10 images and there are only 7 classes with more than 1000 images. For a good model you need at least 1000 images per class.","metadata":{}},{"cell_type":"code","source":"freq_df[freq_df['frequency'] < 10]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"value_counts.index[:10].tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Create a new column with jpg url","metadata":{}},{"cell_type":"code","source":"def jpgurl(df, dir_name='../input/landmark-recognition-2021/train/'):\n    \"\"\"This function will create a url based on the first 3 values of ID\"\"\"\n    for row in range(len(df.index)):\n        df.at[row, 'url'] = os.path.join(dir_name, df['id'][row][0], df['id'][row][1], df['id'][row][2], df['id'][row] + \".\" + 'jpg')\n    return df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = jpgurl(train)\nsample.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.iloc[0][2]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(30,30))\ncolumns = 10\nrows = 10\nfor i in range(1, columns*rows +1):\n    img = PIL.Image.open(sample['url'][i], mode='r')\n    fig.add_subplot(rows, columns, i)\n    plt.imshow(img)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### We find similar images in a database by using transfer learning via a pre-trained VGG-19 image classifier. We retreive the 5 most similar images for each image in the database, and plot the tSNE for all our image feature vectors.","metadata":{}},{"cell_type":"code","source":"!pip install plot_utils","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n sort_utils.py  \n\"\"\"\nimport numpy as np\n\n\"\"\"\n Finds the first unique k elements (based on lowest distance) of lists 'indices' and 'distances'\n\"\"\"\ndef find_topk_unique(indices, distances, k):\n\n    # Sort by ascending distance\n    i_sort_1 = np.argsort(distances)\n    distances_sorted = distances[i_sort_1]\n    indices_sorted = indices[i_sort_1]\n\n    window = np.array(indices_sorted[:k], dtype=int)  # collect first k elements for window intialization\n    window_unique, j_window_unique = np.unique(window, return_index=True)  # find unique window values and indices\n    j = k  # track add index when there are not enough unique values in the window\n    # Run while loop until window_unique has k elements\n    while len(window_unique) != k:\n        # Append new index and value to the window\n        j_window_unique = np.append(j_window_unique, [j])  # append new index\n        window = np.append(window_unique, [indices_sorted[j]])  # append new value\n        # Update the new unique window\n        window_unique, j_window_unique_temp = np.unique(window, return_index=True)\n        j_window_unique = j_window_unique[j_window_unique_temp]\n        # Update add index\n        j += 1\n\n    # Sort the j_window_unique (not sorted) by distances and get corresponding\n    # top-k unique indices and distances (based on smallest distances)\n    distances_sorted_window = distances_sorted[j_window_unique]\n    indices_sorted_window = indices_sorted[j_window_unique]\n    u_sort = np.argsort(distances_sorted_window)  # sort\n\n    distances_top_k_unique = distances_sorted_window[u_sort].reshape((1, -1))\n    indices_top_k_unique = indices_sorted_window[u_sort].reshape((1, -1))\n\n    return indices_top_k_unique, distances_top_k_unique\n\n\"\"\"\n Checks if a list has unique elements\n\"\"\"\ndef is_unique(vec):\n    n_vec = len(vec)\n    n_vec_unique = len(np.unique(vec))\n    return (n_vec == n_vec_unique)\n\n\ndef main():\n    # Example usage\n    indices = np.array([1, 2, 3, 2, 3, 4, 3, 3, 2, 1, 5], dtype=int)\n    distances = np.array([0.8, 0.5, 0.055, 0.4, 0.5, 0.2, 0.1, 0.8, 0.9, 1.0, 0.05], dtype=float)\n\n    n_neighbors = 4\n    indices, distances = find_topk_unique(indices, distances, n_neighbors)\n\n    print(indices)\n    print(distances)\n\n# # Main driver\n# if __name__ == \"main\":\n#     main()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys, os\nimport numpy as np\nfrom keras.preprocessing import image\nfrom keras.models import Model\nsys.path.append(\"src\")\nfrom keras import applications\n# from imagenet_utils import preprocess_input\nfrom tensorflow.keras.applications.vgg19 import VGG19\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.applications.vgg19 import preprocess_input\n# from plot_utils import plot_query_answer\nfrom sklearn.neighbors import KNeighborsClassifier as kNN\nfrom sklearn.manifold import TSNE\n# from TSNE import plot_tsne","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tf.keras.applications.VGG19(\n#     include_top=True,\n#     weights=\"imagenet\",\n#     input_tensor=None,\n#     input_shape=None,\n#     pooling=None,\n#     classes=1000,\n#     classifier_activation=\"softmax\",\n# )","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Loading VGG-19 pre-trained model...\")\nbase_model=applications.vgg19.VGG19(weights='imagenet')\nbase_model.summary()\nmodel = Model(base_model.input,base_model.get_layer('block5_conv4').output) #try extracting from a different layer","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs, filename_heads, X = [], [], []\npath = \"db_new\" \nprint(\"Reading images from '{}' directory...\\n\".format(path))\nfor f in sample[\"url\"]:\n    filename_full = f  # full path filename\n\n    # Read image file\n    img = image.load_img(filename_full, target_size=(224,224))  # resize images as required by the pre-trained model\n    imgs.append(np.array(img))  # image\n    filename_heads.append(f)  # filename head\n\n#     # Pre-process for model input\n#     img = image.img_to_array(img)  # convert to array\n#     img = np.expand_dims(img, axis=0)\n#     img = preprocess_input(img)\n#     features = model.predict(img).flatten()  # features\n#     X.append(features)  # append feature extractor\n\n# X = np.array(X)  # feature vectors\n# imgs = np.array(imgs)  # images\n# print(\"imgs.shape = {}\".format(imgs.shape))\n# print(\"X_features.shape = {}\\n\".format(X.shape))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filename_heads[0:5]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for img in imgs:\n    # Pre-process for model input\n    img = image.img_to_array(img)  # convert to array\n    img = np.expand_dims(img, axis=0)\n    img = preprocess_input(img)\n    features = model.predict(img).flatten()  # features\n    X.append(features)  # append feature extractor\n\n# X = np.array(X)  # feature vectors\n# imgs = np.array(imgs)  # images\n# print(\"imgs.shape = {}\".format(imgs.shape))\n# print(\"X_features.shape = {}\\n\".format(X.shape))","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}