{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<br>\n\n<br><center><img src=\"https://i.ibb.co/zr0kVr8/0-S4-LF1-Obk-Vh2ke-I.jpg\" width=100%></center>\n\n<h2 style=\"text-align: center; font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: underline; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">Google Universal Image Embedding - EDA & BASELINE</h2>\n<h5 style=\"text-align: center; font-family: Verdana; font-size: 12px; font-style: normal; font-weight: bold; text-decoration: None; text-transform: none; letter-spacing: 1px; color: black; background-color: #ffffff;\">CREATED BY: DARIEN SCHETTLER</h5>\n\n<br>\n\n---\n\n<br>\n\n<center><div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">🛑 &nbsp; WARNING:</b><br><br><b>THIS IS A WORK IN PROGRESS</b><br>\n</div></center>\n\n\n<center><div class=\"alert alert-block alert-warning\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">👏 &nbsp; IF YOU FORK THIS OR FIND THIS HELPFUL &nbsp; 👏</b><br><br><b style=\"font-size: 22px; color: darkorange\">PLEASE UPVOTE!</b><br><br>This was a lot of work for me and while it may seem silly, it makes me feel appreciated when others like my work. 😅\n</div></center>\n\n\n","metadata":{}},{"cell_type":"markdown","source":"<p id=\"toc\"></p>\n\n<br><br>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #DA9186; background-color: #ffffff;\">TABLE OF CONTENTS</h1>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#imports\">0&nbsp;&nbsp;&nbsp;&nbsp;IMPORTS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#background_information\">1&nbsp;&nbsp;&nbsp;&nbsp;BACKGROUND INFORMATION</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#setup\">2&nbsp;&nbsp;&nbsp;&nbsp;SETUP</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#helper_functions\">3&nbsp;&nbsp;&nbsp;&nbsp;HELPER FUNCTIONS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#dataset_exploration\">4&nbsp;&nbsp;&nbsp;&nbsp;DATASET EXPLORATION</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#modelling\">5&nbsp;&nbsp;&nbsp;&nbsp;MODELLING</a></h3>\n\n---","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"imports\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #DA9186;\" id=\"imports\">0&nbsp;&nbsp;IMPORTS&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>","metadata":{}},{"cell_type":"code","source":"print(\"\\n... IMPORTS STARTING ...\\n\")\n\nprint(\"\\n\\tVERSION INFORMATION\")\n# Machine Learning and Data Science Imports\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport tensorflow_hub as tfhub; print(f\"\\t\\t– TENSORFLOW HUB VERSION: {tfhub.__version__}\");\nimport tensorflow_addons as tfa; print(f\"\\t\\t– TENSORFLOW ADDONS VERSION: {tfa.__version__}\");\nimport tensorflow_io as tfio; print(f\"\\t\\t– TENSORFLOW I/O VERSION: {tfio.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None;\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\");\nfrom sklearn.preprocessing import RobustScaler, PolynomialFeatures\nfrom pandarallel import pandarallel; pandarallel.initialize();\nfrom sklearn.model_selection import GroupKFold, StratifiedKFold\nfrom scipy.spatial import cKDTree\n\n# RAPIDS\nimport cudf, cupy, cuml\nfrom cuml.neighbors import NearestNeighbors\nfrom cuml.manifold import TSNE, UMAP\nfrom cuml import PCA\n\n# Built In Imports\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom zipfile import ZipFile\nfrom glob import glob\nimport openslide\nimport warnings\nimport requests\nimport hashlib\nimport imageio\nimport IPython\nimport sklearn\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport os\nimport gc\nimport re\n\n# Visualization Imports\nfrom matplotlib.colors import ListedColormap\nfrom matplotlib.patches import Rectangle\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nimport plotly.express as px\nimport tifffile as tif\nimport seaborn as sns\nfrom PIL import Image, ImageEnhance; Image.MAX_IMAGE_PIXELS = 5_000_000_000;\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom matplotlib import animation, rc; rc('animation', html='jshtml')\nimport plotly\nimport PIL\nimport cv2\n\nimport plotly.io as pio\nprint(pio.renderers)\n\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\n    \nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:13:45.960375Z","iopub.execute_input":"2022-07-13T03:13:45.961083Z","iopub.status.idle":"2022-07-13T03:13:58.796085Z","shell.execute_reply.started":"2022-07-13T03:13:45.960974Z","shell.execute_reply":"2022-07-13T03:13:58.795047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<a id=\"background_information\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #DA9186; background-color: #ffffff;\" id=\"background_information\">1&nbsp;&nbsp;BACKGROUND INFORMATION&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>\n\n---\n\n<b><mark>NOTE: This is a very different competition than normal!</mark></b><br><br>\n<b>If you're confused, please read the below details and any additional host provided information to help you understand better</b>","metadata":{}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">1.1 BASIC COMPETITION INFORMATION</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">PRIMARY TASK DESCRIPTION</b>\n\nThe goal of this competition is to develop a computer vision model(s) that are expected to retrieve relevant database images in response to a provided query image (ie, the model should retrieve database images containing the same object as the query).\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">BASIC BACKGROUND INFORMATION</b>\n\nImage representations are a critical building block of computer vision applications. Traditionally, research on image embedding learning has been conducted with a focus on per-domain models. <b>Generally, papers propose generic embedding learning techniques which are applied to different domains separately, rather than developing generic embedding models which could be applied to all domains combined.</b>\n\nThis year's competition is structured in a representation learning format: \n* You will create a model that extracts a feature embedding for the images and submit the model via Kaggle Notebooks. \n* Kaggle will run your model on a held-out test set, perform a k-nearest-neighbors lookup, and score the resulting embedding quality. \n* Both Tensorflow and PyTorch models are supported.\n\nIn this competition, you are asked to develop models that can efficiently retrieve images from a large database. For each query image, the model is expected to retrieve the most similar images from an index set. \n\n<br><b><mark>THERE IS NO TRAINING DATASET</mark></b>. In accordance with the rules, any training data may be used, as long as it is disclosed in the forum by the relevant deadline. There exist many public datasets for different object types (e.g., artworks, landmarks, products, etc), and we encourage participants to experiment with them as needed.\n\nThis competition is part of the <a href=\"https://eccv2022.ecva.net/\"><b>Instance-Level Recognition workshop</b></a> at <a href=\"https://ilr-workshop.github.io/ECCVW2022\"><b>ECCV 2022</b></a>. Top submissions for the competition will be invited to give talks at the workshop. Attending the workshop is not required to participate in the competition, however only teams that are attending the workshop will be considered to present their work.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">COMPETITION HOST INFORMATION</b>\n\n\n<b><a href=\"https://en.wikipedia.org/wiki/Google\">GOOGLE</a></b> LLC is an American multinational technology company that focuses on artificial intelligence, search engine technology, online advertising, cloud computing, computer software, quantum computing, e-commerce, and consumer electronics.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">VISUAL EXPLANATION</b>\n\n<center><img src=\"https://www.biorxiv.org/content/biorxiv/early/2020/06/16/2020.06.15.153247/F1.large.jpg\"></center><br>\n\n<i><b>Figure 1:</b>&nbsp;&nbsp; Schematic of an example instance-prototype contrastive learning framework. It uses a base convolutional neural network $fθ(x)$ to encode each image as a normalized $128-D$ feature vector ($z_i$). The feature embedding is learned by bringing each image projection $z_i$ closer to its instance prototype Embedded Image, and separating it from the previously encountered $z_i$ in the memory queue. <a href=\"https://www.biorxiv.org/content/10.1101/2020.06.15.153247v1.full\">[ref]</a>\n</i><br>\n\n<br>\n\n<center><img src=\"https://i.ibb.co/8DzxhNk/Screen-Shot-2022-07-12-at-1-25-26-PM.png\"></center>\n\n<br>\n\n<i><b>Figure 2:</b>&nbsp;&nbsp; (a) In the latent space, a set of hierarchical prototypes are used to represent the hierarchical semantic structures underlying an image dataset. (b) Instance-wise and prototypical contrastive selective coding select semantically correct positive and negative pairs for contrastive learning, which is guided by the semantic information from hierarchical prototypes. <a href=\"https://www.biorxiv.org/content/10.1101/2020.06.15.153247v1.full\">[ref]</a>\n</i><br>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">1.2 COMPETITION EVALUATION</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">GENERAL EVALUATION INFORMATION</b>\n\nSubmissions are evaluated according to the mean Precision @ 5 metric, where we introduce a small modification to avoid penalizing queries with fewer than 5 expected index images. \n\n<br>\n\nIn detail, the metric is computed as follows:\n\n$$\nmP@5 = \\frac{1}{Q} \\sum_{q=1}^{Q} \\frac{1}{min(n_q, 5)} \\sum_{j=1}^{min(n_q, 5)} rel_q(j)\n$$\n\nWhere: \n* $Q$ is the number of query images\n* $n_q$ is the number of index images containing an object in common with the query image $q$. Note that $n_q > 0$.\n* $rel_q(j)$ denotes the relevance of prediciton $j$ for the $q$-th query: \n    * It will be $1$ if the $j$-th prediction is **correct**\n    * It will be $0$ if the $j$-th prediction is **incorrect**\n\n<br>\n\n---\n\n<br>\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">SUBMISSION FILE INFORMATION</b>\n\n<mark>Unlike a traditional Kaggle code competition where notebooks are rerun top-to-bottom on a private test set... <b>in this competition you will be submitting a model file</b></mark>.\n\n<br>\n\n<b>MODEL REQUIREMENTS:</b>\n* The model must <b>take an image</b> as an input\n* The model must <b>return a float vector</b> (i.e., the image embedding) at the output\n* The embedding (output) <b>dimensionality should be no greater than <mark>64</mark></b>. \n* The model must be <b>packaged into a <code>submission.zip</code> file</b>\n* The model must be <b>compatible with either of the following library versions</b>:\n    * <b>TensorFlow 2.6.4</b>\n    * <b>Pytorch 1.11.0</b>\n\n<br>\n    \n<b>SUBMISSION ZIP FILE REQUIREMENTS:</b>\n* The <b><code>submission.zip</code></b> should contain one (and only one) of the following:\n    * Files and directories in the <b><mark><a href=\"https://www.tensorflow.org/tutorials/keras/save_and_load#savedmodel_format\">Tensorflow's SavedModel format</a></mark></b>.\n    * Files and directories in the <b><mark><a href=\"https://pytorch.org/docs/stable/jit.html\">PyTorch’s TorchScript format</a>, named “saved_model.pt”</mark></b>.\n\n<br>\n\n<b>SUBMISSION PROCESS:</b>\n* Kaggle will use the submitted model to:\n    1. Extract embeddings for the private test and index image sets.\n    2. Create a <b>kNN (k = 5)</b> lookup for each test sample using the <b>Euclidean distance between test and index embeddings</b>.\n    3. Score the quality of the lookups using the competition metric\n\n<br>\n\n<b>EXAMPLE BASELINE KERNELS:</b>\n* To submit your model, all you need is a kernel that produces a <b><code>submission.zip</code></b> file containing the model. \n* Here are examples:\n    * <b>Tensorflow</b>\n        * <b><a href=\"https://www.kaggle.com/andrefaraujo/tf-baseline-submission\">See this kernel.</a></b>\n    * <b>PyTorch</b>\n        * <b><a href=\"https://www.kaggle.com/andrefaraujo/pytorch-baseline-submission\">See this kernel.</a></b>\n\n<br>\n\n<b>BASELINE MODEL IMPLEMENTATION</b>\n* The hosts have graciously provided a <b><a href=\"https://github.com/google-research/google-research/tree/master/universal_embedding_challenge\">Github Repository</a></b> containing code showing how to generate a baseline model for both Tensorflow and Pytorch\n* Here is the link for the general <b><a href=\"https://github.com/google-research/google-research/tree/master/universal_embedding_challenge\">Github Repository</a></b>\n    * <b>https://github.com/google-research/google-research/tree/master/universal_embedding_challenge</b>\n* Here is the link for the respective <b><a href=\"https://www.kaggle.com/datasets/dschettler8845/google-universal-embedding-challenge-github-repo\">Kaggle Dataset</a></b> containing the same files \n    * <b>https://www.kaggle.com/datasets/dschettler8845/google-universal-embedding-challenge-github-repo</b>\n\n<br><font color=\"red\"><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">IS THIS A CODE COMPETITION?</b></font>\n\n<font color=\"red\" style=\"font-size: 30px\"><b>YES</b></font>&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;&nbsp;<b>&#10514; ... To see submission details see the respective section above ... &#10514;</b>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">1.3 DATASET OVERVIEW</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">GENERAL INFORMATION</b>\n\n<br>\n\n---\n\n<b style=\"font-size: 20px; color: darkred;\">THERE IS NO TRAINING DATASET</b>\n\n<br>\n\nI wanted to make that super clear before we move on to speaking about the private dataset and later (updates coming) we will discuss likely public datasets that will be valuable during training.\n\n---\n\n<br>\n\n<b>This is what the hosts have to say about the data:</b>\n\n> In this competition, you are asked to develop models that can efficiently retrieve images from a large database. For each query image, the model is expected to retrieve the most similar images from an index set.<br><br>\n><b>We do not provide a training set</b>. In accordance with the rules, <b>any training data may be used, as long as it is disclosed in the forum by the relevant deadline</b>. There exist <b>many public datasets for different object types (e.g., artworks, landmarks, products, etc)</b>, and we encourage participants to experiment with them as needed.<br><br>\n>Our <b>evaluation dataset, which is kept private, contains images of the following types of object: apparel & accessories, packaged goods, landmarks, furniture & home decor, storefronts, dishes, artwork, toys, memes, illustrations and cars</b>.<br><br>\n><img src=\"https://storage.googleapis.com/kaggle-competitions/kaggle/36414/logos/header.png?t=2022-07-06-20-40-23\">\n> <br>\n> The image above shows examples of images in these domains, which are similar to the ones in the evaluation set. <b>Labels are defined generally at the instance level</b>, for example: the same T-shirt, the same building, the same painting, the same dish.<br><br>\n> The following plot shows the <b>distribution of object types in the dataset</b>:\n> <br><br>\n><img src=\"https://www.googleapis.com/download/storage/v1/b/kaggle-user-content/o/inbox%2F3258%2Fa02c90ad69a049b11a39bd38abcde2cb%2Fdomains.png?generation=1657141626387503&alt=media\">\n> <br>\n\n<br>\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">FILE & DATA FIELD INFORMATION</b>\n\nNo data provided!!!\n\n<br>\n\n<center><img src=\"https://www.meme-arsenal.com/memes/e0ff869187796f8a8a24146b92363adb.jpg\" width=80%></center>\n\n<br>","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"background_information\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #DA9186; background-color: #ffffff;\" id=\"setup\">2&nbsp;&nbsp;SETUP&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>\n\n---\n","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">2.1 ACCELERATOR DETECTION</h3>\n\n---\n\nIn order to use **`TPU`**, we use **`TPUClusterResolver`** for the initialization which is necessary to connect to the remote cluster and initialize cloud TPUs. Let's go over two important points\n\n1. When using TPU on Kaggle, you don't need to specify arguments for **`TPUClusterResolver`**\n2. However, on **G**oogle **C**ompute **E**ngine (**GCE**), you will need to do the following:\n\n<br>\n\n```python\n# The name you gave to the TPU to use\nTPU_WORKER = 'my-tpu-name'\n\n# or you can also specify the grpc path directly\n# TPU_WORKER = 'grpc://xxx.xxx.xxx.xxx:8470'\n\n# The zone you chose when you created the TPU to use on GCP.\nZONE = 'us-east1-b'\n\n# The name of the GCP project where you created the TPU to use on GCP.\nPROJECT = 'my-tpu-project'\n\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver(tpu=TPU_WORKER, zone=ZONE, project=PROJECT)\n```\n\n<div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">🛑 &nbsp; WARNING:</b><br><br>- Although the Tensorflow documentation says it is the <b>project name</b> that should be provided for the argument <b><code>`project`</code></b>, it is actually the <b>Project ID</b>, that you should provide. This can be found on the GCP project dashboard page.<br>\n</div>\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📖 &nbsp; REFERENCES:</b><br><br>\n    - <a href=\"https://www.tensorflow.org/guide/tpu#tpu_initialization\"><b>Guide - Use TPUs</b></a><br>\n    - <a href=\"https://www.tensorflow.org/api_docs/python/tf/distribute/cluster_resolver/TPUClusterResolver\"><b>Doc - TPUClusterResolver</b></a><br>\n\n</div>","metadata":{}},{"cell_type":"code","source":"print(f\"\\n... ACCELERATOR SETUP STARTING ...\\n\")\n\n# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    TPU = tf.distribute.cluster_resolver.TPUClusterResolver()  \nexcept ValueError:\n    TPU = None\n\nif TPU:\n    print(f\"\\n... RUNNING ON TPU - {TPU.master()}...\")\n    tf.config.experimental_connect_to_cluster(TPU)\n    tf.tpu.experimental.initialize_tpu_system(TPU)\n    strategy = tf.distribute.experimental.TPUStrategy(TPU)\nelse:\n    print(f\"\\n... RUNNING ON CPU/GPU ...\")\n    # Yield the default distribution strategy in Tensorflow\n    #   --> Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n    tf.config.experimental.set_memory_growth(tf.config.list_physical_devices('GPU')[0], True)\n\n# What Is a Replica?\n#    --> A single Cloud TPU device consists of FOUR chips, each of which has TWO TPU cores. \n#    --> Therefore, for efficient utilization of Cloud TPU, a program should make use of each of the EIGHT (4x2) cores. \n#    --> Each replica is essentially a copy of the training graph that is run on each core and \n#        trains a mini-batch containing 1/8th of the overall batch size\nN_REPLICAS = strategy.num_replicas_in_sync\n    \nprint(f\"... # OF REPLICAS: {N_REPLICAS} ...\\n\")\n\nprint(f\"\\n... ACCELERATOR SETUP COMPLTED ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:13:58.798185Z","iopub.execute_input":"2022-07-13T03:13:58.798512Z","iopub.status.idle":"2022-07-13T03:13:58.826519Z","shell.execute_reply.started":"2022-07-13T03:13:58.798478Z","shell.execute_reply":"2022-07-13T03:13:58.825589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">2.2 COMPETITION DATA ACCESS</h3>\n\n---\n\nTPUs read data must be read directly from **G**oogle **C**loud **S**torage **(GCS)**. Kaggle provides a utility library – **`KaggleDatasets`** – which has a utility function **`.get_gcs_path`** that will allow us to access the location of our input datasets within **GCS**.<br><br>\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📌 &nbsp; TIPS:</b><br><br>- If you have multiple datasets attached to the notebook, you should pass the name of a specific dataset to the <b><code>`get_gcs_path()`</code></b> function. <i>In our case, the name of the dataset is the name of the directory the dataset is mounted within.</i><br><br>\n</div>","metadata":{}},{"cell_type":"code","source":"print(\"\\n... DATA ACCESS SETUP STARTED ...\\n\")\n\nDEMO_DIR = \"/kaggle/input/google-universal-embedding-challenge-github-repo/universal_embedding_challenge\"\nif TPU:\n    MODEL_DIR = os.path.join(KaggleDatasets().get_gcs_path('tf-efficientnetv2-2022-models-b0b3'), \"tf_keras_applications\")\n    EVAL_DIR = os.path.join(KaggleDatasets().get_gcs_path('caltech256'), \"256_ObjectCategories\")\n    save_locally = tf.saved_model.SaveOptions(experimental_io_device='/job:localhost')\n    load_locally = tf.saved_model.LoadOptions(experimental_io_device='/job:localhost')\nelse:\n    EVAL_DIR = \"/kaggle/input/caltech256/256_ObjectCategories\"\n    MODEL_DIR = \"/kaggle/input/tf-efficientnetv2-2022-models-b0b3/tf_keras_applications\"\n    save_locally = None\n    load_locally = None\n\nprint(f\"\\n... MODEL DIRECTORY PATH IS:\\n\\t--> {MODEL_DIR}\")\nprint(f\"... EVAL DATA DIRECTORY PATH IS:\\n\\t--> {EVAL_DIR}\\n\")\n\nprint(f\"\\n... IMMEDIATE CONTENTS OF MODEL DIRECTORY IS:\")\nfor file in tf.io.gfile.glob(os.path.join(MODEL_DIR, \"*\")): print(f\"\\t--> {file}\")\n\nprint(f\"\\n... IMMEDIATE CONTENTS OF EVAL DATA DIRECTORY IS:\")\nfor file in tf.io.gfile.glob(os.path.join(EVAL_DIR, \"*\")): print(f\"\\t--> {file}\")\n\n    \nprint(\"\\n\\n... DATA ACCESS SETUP COMPLETED ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:13:58.829500Z","iopub.execute_input":"2022-07-13T03:13:58.829801Z","iopub.status.idle":"2022-07-13T03:13:58.910966Z","shell.execute_reply.started":"2022-07-13T03:13:58.829777Z","shell.execute_reply":"2022-07-13T03:13:58.909591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">2.3 LEVERAGING XLA OPTIMIZATIONS</h3>\n\n---\n\n\n**XLA** (Accelerated Linear Algebra) is a domain-specific compiler for linear algebra that can accelerate TensorFlow models with potentially no source code changes. **The results are improvements in speed and memory usage**.\n\n<br>\n\nWhen a TensorFlow program is run, all of the operations are executed individually by the TensorFlow executor. Each TensorFlow operation has a precompiled GPU/TPU kernel implementation that the executor dispatches to.\n\nXLA provides us with an alternative mode of running models: it compiles the TensorFlow graph into a sequence of computation kernels generated specifically for the given model. Because these kernels are unique to the model, they can exploit model-specific information for optimization.<br><br>\n\n<div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">🛑 &nbsp; WARNING:</b><br><br>- XLA can not currently compile functions where dimensions are not inferrable: that is, if it's not possible to infer the dimensions of all tensors without running the entire computation<br>\n</div>\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📌 &nbsp; NOTE:</b><br><br>- XLA compilation is only applied to code that is compiled into a graph (in <b>TF2</b> that's only a code inside <b><code>tf.function</code></b>).<br>- The <b><code>jit_compile</code></b> API has must-compile semantics, i.e. either the entire function is compiled with XLA, or an <b><code>errors.InvalidArgumentError</code></b> exception is thrown)\n</div>\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📖 &nbsp; REFERENCE:</b><br><br>    - <a href=\"https://www.tensorflow.org/xla\"><b>XLA: Optimizing Compiler for Machine Learning</b></a><br>\n</div>","metadata":{}},{"cell_type":"code","source":"print(f\"\\n... XLA OPTIMIZATIONS STARTING ...\\n\")\n\nprint(f\"\\n... CONFIGURE JIT (JUST IN TIME) COMPILATION ...\\n\")\n# enable XLA optmizations (10% speedup when using @tf.function calls)\ntf.config.optimizer.set_jit(True)\n\nprint(f\"\\n... XLA OPTIMIZATIONS COMPLETED ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:13:58.917155Z","iopub.execute_input":"2022-07-13T03:13:58.919172Z","iopub.status.idle":"2022-07-13T03:13:58.929006Z","shell.execute_reply.started":"2022-07-13T03:13:58.919126Z","shell.execute_reply":"2022-07-13T03:13:58.928073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">2.4 BASIC DATA DEFINITIONS & INITIALIZATIONS</h3>\n\n---\n\nWe load an evaluation dataset to test our approach (caltech256)","metadata":{}},{"cell_type":"code","source":"print(\"\\n... BASIC DATA SETUP STARTING ...\\n\\n\")\n\neval_df = pd.DataFrame({\"img_path\":tf.io.gfile.glob(os.path.join(EVAL_DIR, \"**\", \"*.jpg\"))})\n\nADD_SHAPE_INFO=False\nif ADD_SHAPE_INFO:\n    eval_df[\"img_shape\"] = eval_df[\"img_path\"].progress_apply(lambda x: Image.open(x).size)\n\neval_df[\"label\"] = eval_df.img_path.apply(lambda x: x.rsplit(\"/\", 2)[1])\neval_df[\"label_str\"] = eval_df[\"label\"].apply(lambda x: x.split(\".\")[1])\neval_df[\"label_int\"] = eval_df[\"label\"].apply(lambda x: x.split(\".\")[0]).astype(int)\neval_df[\"label_int_0_offset\"] = eval_df.label_int-1\n\ndisplay(eval_df)\n\nfig = px.histogram(eval_df, \"label\", color=\"label\", title=\"<b>CalTech256 Class Distribution</b>\")\nfig.update_layout(showlegend=False)\nfig.show()\n\nN_CLASSES = eval_df.label.nunique()\nS2I_CLASS_MAP = eval_df[[\"label_str\", \"label_int_0_offset\"]].groupby(\"label_str\").first()[\"label_int_0_offset\"].to_dict()\nI2S_CLASS_MAP = {v:k for k,v in S2I_CLASS_MAP.items()}\n\nprint(\"\\n... BASIC DATA SETUP FINISHED ...\\n\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:13:58.930685Z","iopub.execute_input":"2022-07-13T03:13:58.932255Z","iopub.status.idle":"2022-07-13T03:14:06.001887Z","shell.execute_reply.started":"2022-07-13T03:13:58.932220Z","shell.execute_reply":"2022-07-13T03:14:06.000993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n\n<a id=\"helper_functions\"></a>\n\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #DA9186; background-color: #ffffff;\" id=\"helper_functions\">\n    3&nbsp;&nbsp;HELPER FUNCTION & CLASSES&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a>\n</h1>\n\n---\n\nI've created some helper functions specifically for this competition to deal with the more unwieldy images. I tried my best to document everything thoroughly but if you need help please leave a comment!","metadata":{}},{"cell_type":"code","source":"def flatten_l_o_l(nested_list):\n    \"\"\" Flatten a list of lists \"\"\"\n    return [item for sublist in nested_list for item in sublist]\n\ndef print_ln(symbol=\"-\", line_len=110):\n    print(symbol*line_len)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:14:06.003191Z","iopub.execute_input":"2022-07-13T03:14:06.004080Z","iopub.status.idle":"2022-07-13T03:14:06.010432Z","shell.execute_reply.started":"2022-07-13T03:14:06.004045Z","shell.execute_reply":"2022-07-13T03:14:06.009064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n\n<a id=\"dataset_exploration\"></a>\n\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #DA9186; background-color: #ffffff;\" id=\"dataset_exploration\">\n    4&nbsp;&nbsp;DATASET EXPLORATION&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a>\n</h1>\n\n---\n\nTo be done later...","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">4.1 PUBLIC DATASET #1</h3>\n\n---\n\n<b>PLACEHOLDER WIP PLACEHOLDER</b>\n\n* CALTECH256\n* ObjectNet\n* ImageNet\n* PascalVOC\n* COCO\n","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">4.1 PUBLIC DATASET #2</h3>\n\n---\n\n<b>PLACEHOLDER WIP PLACEHOLDER</b>\n\n* CALTECH256\n* ObjectNet\n* ImageNet\n* PascalVOC\n* COCO\n","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"modelling\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #DA9186;\" id=\"modelling\">5&nbsp;&nbsp;MODELLING&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>\n\nFor this baseline approach we will be using a pretrained, headless **`EfficientNetV2b0`** model that we will modify to reduce the dimensionality to only 32 dims.","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">5.1 LOAD EFFICIENTNETV2 MODEL</h3>\n\n---\n\nWe've created a repository containing some **EfficientNetV2** models so that we can load them offline as they would normally need to be downloaded.","metadata":{}},{"cell_type":"code","source":"def tf_pad_to_square(a, constant=255):\n    \"\"\" Pad a tensor array `a` evenly until it is a square \"\"\"\n    h_src = tf.shape(a)[0]\n    w_src = tf.shape(a)[1] \n    if w_src>h_src: # pad height\n        n_to_add = w_src-h_src\n        top_pad = n_to_add//2\n        bottom_pad = n_to_add-top_pad\n        a = tf.pad(a, [(top_pad, bottom_pad), (0, 0), (0, 0)], mode='constant', constant_values=constant)\n    elif h_src>w_src: # pad width\n        n_to_add = h_src-w_src\n        left_pad = n_to_add//2\n        right_pad = n_to_add-left_pad\n        a = tf.pad(a, [(0, 0), (left_pad, right_pad), (0, 0)], mode='constant', constant_values=constant)\n    else:\n        pass\n    return a\n\ndef tf_load_image(img_path, resize_to=(224,224), pad_to_square=True, no_prep=True):\n    \"\"\" Load an image with the correct size and shape \"\"\"\n    img = tf.io.decode_image(tf.io.read_file(img_path), channels=3, expand_animations=False)\n    if not no_prep: img = tf.image.resize_with_pad(img, tf_input_shape_0, tf_input_shape_1)\n    return img\n\ndef add_preprocessing_to_model(_model):\n    _inputs = tf.keras.layers.Input(shape=(None, None, 3), name=\"inputs\", dtype=tf.uint8)\n    x = tf.keras.layers.Lambda(\n        lambda _x: tf.image.resize_with_pad(_x, tf_input_shape_0, tf_input_shape_1)\n    )(_inputs)\n    _outputs = _model(x)\n    return tf.keras.Model(inputs=_inputs, outputs=_outputs)\n    \n_ev2_input_shape_map = {\"efficientnetv2_b0\":224, \"efficientnetv2_b1\":240, \"efficientnetv2_b2\":260, \"efficientnetv2_b3\":300,}\nMODEL_NAME, MODEL_LEVEL = \"efficientnetv2\", \"b0\"\nMODEL_STR = f\"{MODEL_NAME}_{MODEL_LEVEL}\"\nMODEL_PATH, INPUT_SHAPE = os.path.join(MODEL_DIR, MODEL_STR), (_ev2_input_shape_map[MODEL_STR], _ev2_input_shape_map[MODEL_STR])\ntf_input_shape_0, tf_input_shape_1 = tf.constant(INPUT_SHAPE[0]), tf.constant(INPUT_SHAPE[1])\nmodel = tf.keras.models.load_model(MODEL_PATH)\n\nprint(f\"\\n... MODEL SUMMARY FOR {MODEL_STR.upper()}  ({INPUT_SHAPE}) ...\\n\\n\")\nmodel.summary()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-13T03:14:08.216675Z","iopub.execute_input":"2022-07-13T03:14:08.217708Z","iopub.status.idle":"2022-07-13T03:14:31.761994Z","shell.execute_reply.started":"2022-07-13T03:14:08.217674Z","shell.execute_reply":"2022-07-13T03:14:31.760986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">5.2 USE FULL EMBEDDING AND EVAL DATASET TO EXPERIMENT</h3>\n\n---\n\nOur current model outputs an embedding that is much too large. That being said, let's see how good the entire embedding is at classifying our evaluation dataset before we further reduce the dimensionality.\n\nThe cells below will follow these steps:\n\n1. Create dataset to allow for easy prediction with tf model\n2. Generate embeddings for entire evaluation dataset\n3. Find Similar Images with RAPIDS KNN","metadata":{}},{"cell_type":"markdown","source":"<br>\n\n**STEP 1:**\n* Create dataset to allow for easy prediction with tf model","metadata":{}},{"cell_type":"code","source":"# HYPERPARAMETERS\nBATCH_SIZE = 64\neval_ds = tf.data.Dataset.from_tensor_slices((eval_df.img_path.values, eval_df.label_int.values))\neval_ds = eval_ds.map(lambda x,y: (tf_load_image(x, INPUT_SHAPE, no_prep=False), tf.one_hot(y, N_CLASSES)), num_parallel_calls=tf.data.AUTOTUNE)\\\n                 .batch(BATCH_SIZE, num_parallel_calls=tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:14:31.764033Z","iopub.execute_input":"2022-07-13T03:14:31.764627Z","iopub.status.idle":"2022-07-13T03:14:31.953263Z","shell.execute_reply.started":"2022-07-13T03:14:31.764585Z","shell.execute_reply":"2022-07-13T03:14:31.952287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n**STEP 2:**\n* Generate embeddings for entire evaluation dataset","metadata":{}},{"cell_type":"code","source":"eval_image_path = eval_df[\"img_path\"].values\neval_labels = eval_df[\"label_int_0_offset\"].values\neval_embeddings = model.predict(eval_ds, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:14:31.954569Z","iopub.execute_input":"2022-07-13T03:14:31.954947Z","iopub.status.idle":"2022-07-13T03:16:44.394936Z","shell.execute_reply.started":"2022-07-13T03:14:31.954896Z","shell.execute_reply":"2022-07-13T03:16:44.393770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n**STEP 3:**\n* Find Similar Images with RAPIDS KNN\n\n<b><mark>Inspired heavily by <a href=\"https://www.kaggle.com/code/cdeotte/rapids-cuml-tfidfvectorizer-and-knn\">this notebook</a> from Chris Deotte</mark></b>","metadata":{}},{"cell_type":"code","source":"# How many neighbors to identify (identity + closest)(ie. real neighbors is -1)\nKNN = 10\nknn_model = NearestNeighbors(n_neighbors=KNN)\nknn_model.fit(eval_embeddings)\ndistances, indices = knn_model.kneighbors(eval_embeddings)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:16:44.398369Z","iopub.execute_input":"2022-07-13T03:16:44.398753Z","iopub.status.idle":"2022-07-13T03:16:46.021698Z","shell.execute_reply.started":"2022-07-13T03:16:44.398715Z","shell.execute_reply":"2022-07-13T03:16:46.020486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def visualize_embedding_similarity(df, idx, _distances, _indices, plot_images=True):\n    print(\"\\n... VISUALIZING SIMILAR EMBEDDING PREDICTIONS FOR THIS EXAMPLE ROW ...\\n\")\n    display(df.iloc[idx:(idx+1)])\n    \n    print(\"\\n... RETRIEVING DISTANCES, CLOSEST INDICES, LABELS AND GROUND TRUTH LABEL ...\\n\")\n    demo_dist, demo_ids = _distances[idx], _indices[idx]\n    demo_img_paths = df.iloc[demo_ids].img_path.values\n    demo_int_lbls, demo_str_lbls = df.iloc[demo_ids].label_int.values, df.iloc[demo_ids].label_str.values\n    gt_demo_int_lbl, gt_demo_str_lbl = demo_int_lbls[0], demo_str_lbls[0]\n    \n    print(\"\\n\\n... PLOTTING LINE/SCATTER TO SHOW NEIGHBOR DISTANCES RELATIVE TO EXAMPLE ROW ...\\n\")\n    fig = px.line(x=[str(x) for x in demo_ids], y=demo_dist, markers=True,\n                 labels={\"x\":\"<b>Row ID of Neighbor</b>\", \"y\":\"<b>Distance From Ground Truth Image</b>\"},\n                 hover_data={\"SAME LABEL AS GT\":[demo_lbl==gt_demo_int_lbl for demo_lbl in demo_int_lbls], \n                             \"ACTUAL INTEGER LABEL\":demo_int_lbls,\n                             \"ACTUAL STRING LABEL\":demo_str_lbls},\n                 title=f\"<b>Example For Row {idx} in the Eval Dataframe</b>\")\n    fig.show()\n    \n    if plot_images:\n        print(\"\\n\\n... PLOTTING ORIGINAL AND 9 CLOSEST IMAGES ...\\n\")\n        plt.figure(figsize=(20,10))\n        for i, img_path in enumerate(demo_img_paths):\n            plt.subplot(2,5,i+1)\n            plt.axis(False)\n            plt.title(\"Example/Seed Image\" if i==0 else f\"#{i} Closest Neighbor (Δ={demo_dist[i]:.2f})\", fontweight=\"bold\")\n            plt.imshow(tf_load_image(img_path).numpy()/255.)\n        plt.tight_layout()\n        plt.show()\n        \n    print(\"\\n\\n... DATAFRAME SHOWING THE EXAMPLE/SEED ROW AND 9 CLOSEST NEIGHBORS ...\\n\")\n    display(df.iloc[demo_ids])\n    \nN_EX = 10\n\nprint_ln()\nprint_ln()\nprint(f\"\\n\\n... DISPLAYING SIMILARITY VISUALIZATION FOR {N_EX} RANDOM EXAMPLES FROM OUR EVAL DATAFRAME ...\\n\\n\")\nprint_ln()\nprint_ln()\n\nrandom_ids = random.sample(range(len(eval_df)), N_EX)\nfor _i in random_ids:\n    print(\"\\n\\n\")\n    print_ln()\n    print(f\"\\n... EXAMPLE #{_i+1} ...\\n\")\n    print_ln()\n    print(\"\\n\")\n    visualize_embedding_similarity(eval_df, idx=_i, _distances=distances, _indices=indices)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:16:46.023417Z","iopub.execute_input":"2022-07-13T03:16:46.023762Z","iopub.status.idle":"2022-07-13T03:16:59.178670Z","shell.execute_reply.started":"2022-07-13T03:16:46.023727Z","shell.execute_reply":"2022-07-13T03:16:59.177675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">5.3 USE PCA TO REDUCE DIMENSIONALITY TO 32</h3>\n\n---\n","metadata":{}},{"cell_type":"code","source":"N_DIM=32\npca_float = PCA(n_components=N_DIM)\npca_float.fit(eval_embeddings)\neval_embeddings_redux = np.dot(eval_embeddings, pca_float.components_.T)\nprint(eval_embeddings_redux.shape)\n\n# How many neighbors to identify (identity + closest)(ie. real neighbors is -1)\nredux_knn_model = NearestNeighbors(n_neighbors=KNN)\nredux_knn_model.fit(eval_embeddings_redux)\nredux_distances, redux_indices = redux_knn_model.kneighbors(eval_embeddings_redux)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:16:59.179889Z","iopub.execute_input":"2022-07-13T03:16:59.180438Z","iopub.status.idle":"2022-07-13T03:17:02.202274Z","shell.execute_reply.started":"2022-07-13T03:16:59.180403Z","shell.execute_reply":"2022-07-13T03:17:02.201325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">5.4 USE REDUX EMBEDDING AND EVAL DATASET TO EXPERIMENT</h3>\n\n---\n\nLet's see how our redux model compares to our original!","metadata":{}},{"cell_type":"code","source":"for _i in random_ids:\n    print(\"\\n\\n\")\n    print_ln()\n    print(f\"\\n... EXAMPLE #{_i+1} ...\\n\")\n    print_ln()\n    print(\"\\n\")\n    visualize_embedding_similarity(eval_df, idx=_i, _distances=redux_distances, _indices=redux_indices)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:17:02.203717Z","iopub.execute_input":"2022-07-13T03:17:02.204117Z","iopub.status.idle":"2022-07-13T03:17:14.376350Z","shell.execute_reply.started":"2022-07-13T03:17:02.204075Z","shell.execute_reply":"2022-07-13T03:17:14.375502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">5.5 UPDATE MODEL WITH DIMENSIONALITY REDUCTION FOR TEST TIME</h3>\n\n---\n","metadata":{}},{"cell_type":"code","source":"tf_pca_comp = tf.constant(pca_float.components_.T)\ndef get_updated_model(_model):\n    _model = add_preprocessing_to_model(_model)\n    _inputs = _model.input\n    _output_1 = tf.keras.layers.Lambda(lambda _x: tf.tensordot(_x, tf_pca_comp, 1), name=\"embedding\")(_model.output)\n    _output_2 = tf.keras.layers.Lambda(lambda _x: tf.nn.l2_normalize(_x), name=\"embedding_norm\")(_output_1)\n    return tf.keras.Model(inputs=_inputs, outputs=[_output_1, _output_2])\n\nup_model = get_updated_model(model)\nup_model.compile()\nup_model.save(f\"updated_{MODEL_STR}_model__{N_DIM}\")\nup_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:17:25.097867Z","iopub.execute_input":"2022-07-13T03:17:25.098283Z","iopub.status.idle":"2022-07-13T03:18:12.802264Z","shell.execute_reply.started":"2022-07-13T03:17:25.098245Z","shell.execute_reply":"2022-07-13T03:18:12.801190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">5.6 CHECK MODEL WILL INFER CORRECTLY FROM SAVED FILE</h3>\n\n---\n","metadata":{}},{"cell_type":"code","source":"# Load the model\n__model = tf.saved_model.load(f\"updated_{MODEL_STR}_model__{N_DIM}\")\n\n# Get the embedding fn\nembedding_fn = __model.signatures[\"serving_default\"]\n\n# Load image 1 and extract its embedding.\nimage_tensor = tf.convert_to_tensor(\n        np.array(Image.open(eval_df.img_path[0]).convert(\"RGB\"))\n    )\nexpanded_tensor = tf.expand_dims(image_tensor, axis=0)\nembedding = embedding_fn(expanded_tensor)[\"embedding_norm\"]\n\n# Display 1\nprint(embedding.shape, embedding)\n\n# Load image 2 and extract its embedding.\nimage_tensor = tf.convert_to_tensor(\n        np.array(Image.open(eval_df.img_path[1]).convert(\"RGB\"))\n    )\nexpanded_tensor = tf.expand_dims(image_tensor, axis=0)\nembedding = embedding_fn(expanded_tensor)[\"embedding_norm\"]\n\n# Display 2\nprint(embedding.shape, embedding)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:18:12.803819Z","iopub.execute_input":"2022-07-13T03:18:12.804432Z","iopub.status.idle":"2022-07-13T03:18:32.778849Z","shell.execute_reply.started":"2022-07-13T03:18:12.804394Z","shell.execute_reply":"2022-07-13T03:18:32.777830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">5.7 ZIP MODEL FOR SUBMISSION</h3>\n\n---\n","metadata":{}},{"cell_type":"code","source":"with ZipFile('submission.zip','w') as zip:    \n    zip.write('/kaggle/working/updated_efficientnetv2_b0_model__32/saved_model.pb', arcname='saved_model.pb') \n    zip.write('/kaggle/working/updated_efficientnetv2_b0_model__32/variables/variables.data-00000-of-00001', arcname='variables/variables.data-00000-of-00001') \n    zip.write('/kaggle/working/updated_efficientnetv2_b0_model__32/variables/variables.index', arcname='variables/variables.index') \n    # zip.write('/kaggle/working/updated_efficientnetv2_b0_model__32/keras_metadata.pb', arcname='keras_metadata.pb') ","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:20:43.267937Z","iopub.execute_input":"2022-07-13T03:20:43.268673Z","iopub.status.idle":"2022-07-13T03:20:43.360301Z","shell.execute_reply.started":"2022-07-13T03:20:43.268637Z","shell.execute_reply":"2022-07-13T03:20:43.359337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #DA9186; background-color: #ffffff;\">5.8 CLEANUP</h3>\n\n---\n","metadata":{}},{"cell_type":"code","source":"!rm -rf ./google*\n!rm -rf ./pca*\n!rm -rf ./updated_efficientnetv2*","metadata":{"execution":{"iopub.status.busy":"2022-07-13T03:20:55.293733Z","iopub.execute_input":"2022-07-13T03:20:55.294100Z","iopub.status.idle":"2022-07-13T03:20:57.568660Z","shell.execute_reply.started":"2022-07-13T03:20:55.294071Z","shell.execute_reply":"2022-07-13T03:20:57.567345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}