{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<br>\n\n<h2 style=\"text-align: center; font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: underline; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">Happy Whale and Dolphin <span style=\"font-size: 16px;\">(😊🐳&🐬)</span><br><br>Train Individual Model<br></h2>\n<h5 style=\"text-align: center; font-family: Verdana; font-size: 12px; font-style: normal; font-weight: bold; text-decoration: None; text-transform: none; letter-spacing: 1px; color: black; background-color: #ffffff;\">CREATED BY: DARIEN SCHETTLER</h5>\n\n<br>\n\n---\n\n<br>\n\n<center><div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">🛑 &nbsp; WARNING:</b><br><br><b>THIS IS A WORK IN PROGRESS</b><br>\n</div></center>\n\n\n<center><div class=\"alert alert-block alert-warning\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">👏 &nbsp; IF YOU FORK THIS OR FIND THIS HELPFUL &nbsp; 👏</b><br><br><b style=\"font-size: 22px; color: darkorange\">PLEASE UPVOTE!</b><br><br>This was a lot of work for me and while it may seem silly, it makes me feel appreciated when others like my work. 😅\n</div></center>\n\n\n","metadata":{"papermill":{"duration":0.120561,"end_time":"2021-11-06T21:15:09.611563","exception":false,"start_time":"2021-11-06T21:15:09.491002","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<p id=\"toc\"></p>\n\n<br><br>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #3eb489; background-color: #ffffff;\">TABLE OF CONTENTS</h1>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#imports\">0&nbsp;&nbsp;&nbsp;&nbsp;IMPORTS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#background_information\">1&nbsp;&nbsp;&nbsp;&nbsp;BACKGROUND INFORMATION</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#setup\">2&nbsp;&nbsp;&nbsp;&nbsp;SETUP</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#helper_functions\">3&nbsp;&nbsp;&nbsp;&nbsp;HELPER FUNCTIONS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#dataset_exploration\">4&nbsp;&nbsp;&nbsp;&nbsp;DATASET EXPLORATION</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#model_baseline\">5&nbsp;&nbsp;&nbsp;&nbsp;BASELINE</a></h3>\n\n---","metadata":{"papermill":{"duration":0.085591,"end_time":"2021-11-06T21:15:09.78303","exception":false,"start_time":"2021-11-06T21:15:09.697439","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"imports\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #3eb489;\" id=\"imports\">0&nbsp;&nbsp;IMPORTS&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>","metadata":{"papermill":{"duration":0.050527,"end_time":"2021-11-06T21:15:09.894476","exception":false,"start_time":"2021-11-06T21:15:09.843949","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"\\n... IMPORTS STARTING ...\\n\")\n\nprint(\"\\n... PIP/APT INSTALLS AND DOWNLOADS/ZIP STARTING ...\")\n# !pip install -q --upgrade tensorflow\n!pip install -q efficientnet\nprint(\"... PIP/APT INSTALLS COMPLETE ...\\n\")\n\nprint(\"\\n\\tVERSION INFORMATION\")\n# Machine Learning and Data Science Imports\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport tensorflow_addons as tfa; print(f\"\\t\\t– TENSORFLOW ADDONS VERSION: {tfa.__version__}\");\nimport tensorflow_hub as tfhub; print(f\"\\t\\t– TENSORFLOW HUB VERSION: {tfhub.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None;\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\");\nfrom sklearn.preprocessing import RobustScaler, PolynomialFeatures\nfrom pandarallel import pandarallel; pandarallel.initialize();\nfrom sklearn.model_selection import GroupKFold;\nimport efficientnet.tfkeras as efn\n\n# RAPIDS\n# import cudf, cupy, cuml\n\n# Built In Imports\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom glob import glob\nimport warnings\nimport requests\nimport hashlib\nimport imageio\nimport IPython\nimport sklearn\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport os\nimport gc\nimport re\n\n# Visualization Imports\nfrom matplotlib.colors import ListedColormap\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nimport plotly.express as px\nimport seaborn as sns\nfrom PIL import Image, ImageEnhance\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom matplotlib import animation, rc; rc('animation', html='jshtml')\nimport plotly\nimport PIL\nimport cv2\n\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\n    \nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{"papermill":{"duration":162.144149,"end_time":"2021-11-06T21:17:52.087371","exception":false,"start_time":"2021-11-06T21:15:09.943222","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-02-23T14:29:22.525748Z","iopub.execute_input":"2022-02-23T14:29:22.526127Z","iopub.status.idle":"2022-02-23T14:29:35.556367Z","shell.execute_reply.started":"2022-02-23T14:29:22.526004Z","shell.execute_reply":"2022-02-23T14:29:35.555218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<a id=\"background_information\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #3eb489; background-color: #ffffff;\" id=\"background_information\">1&nbsp;&nbsp;BACKGROUND INFORMATION&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>\n\n---\n","metadata":{"papermill":{"duration":0.05019,"end_time":"2021-11-06T21:17:52.231372","exception":false,"start_time":"2021-11-06T21:17:52.181182","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">1.1 BASIC COMPETITION INFORMATION</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">PRIMARY TASK DESCRIPTION</b>\n\nIn this competition, you’ll develop a model to **match** individual whales and dolphins by **unique—but often subtle—characteristics** of their natural markings. \n\nYou'll pay particular attention to **dorsal fins** and **lateral body views** in image sets from a **multi-species dataset** built by 28 research institutions.\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">CONTEXT</b>\n\nWe use fingerprints and facial recognition to identify people, but can we use similar approaches with animals? In fact, researchers manually track marine life by the shape and markings on their tails, dorsal fins, heads and other body parts. Identification by natural markings via photographs—known as photo-ID—is a powerful tool for marine mammal science. It allows individual animals to be tracked over time and enables assessments of population status and trends. With your help to automate whale and dolphin photo-ID, researchers can reduce image identification times by over 99%. More efficient identification could enable a scale of study previously unaffordable or impossible.\n\nCurrently, most research institutions rely on time-intensive—and sometimes inaccurate—manual matching by the human eye. Thousands of hours go into manual matching, which involves staring at photos to compare one individual to another, finding matches, and identifying new individuals. While researchers enjoy looking at a whale photo or two, manual matching limits the scope and reach.\n\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">MORE BACKGROUND INFORMATION</b>\n\n\nTBD","metadata":{"papermill":{"duration":0.053372,"end_time":"2021-11-06T21:17:52.337029","exception":false,"start_time":"2021-11-06T21:17:52.283657","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">1.2 COMPETITION EVALUATION</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">GENERAL EVALUATION INFORMATION</b>\n\n**Submissions are evaluated according to the Mean Average Precision @ 5 (MAP@5)**\n\nLATEX TO BE INSERTED\n\n*where 𝑈 is the number of images, 𝑃(𝑘) is the precision at cutoff 𝑘, 𝑛 is the number predictions per image, and 𝑟𝑒𝑙(𝑘) is an indicator function equaling 1 if the item at rank 𝑘 is a relevant (correct) label, zero otherwise.*\n\nOnce a correct label has been scored for an observation, that label is no longer considered relevant for that observation, and additional predictions of that label are skipped in the calculation. For example, if the correct label is A for an observation, the following predictions all score an average precision of 1.0.\n\n```\n[A, B, C, D, E]\n[A, A, A, A, A]\n[A, B, A, C, A]\n```\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">SUBMISSION FORMAT</b>\n\nFor each image in the test set, you may predict up to 5 **`individual_id`** labels. There are individuals in the test set that are not seen in the training data; these should be predicted as new_individual. The file should contain a header and have the following format:\n\n```\nimage,predictions \n000188a72f2562.jpg,37c7aba965a5 114207cab555 a6e325d8e924 19fbb960f07d new_individual \n000ba09273d6f3.jpg,37c7aba965a5 114207cab555 a6e325d8e924 19fbb960f07d new_individual \n...\n``` \n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase; color: red;\">IS THIS A CODE COMPETITION?</b>\n\n<font style=\"color:red; font-weight: bold; font-size: 20px;\">NO!</font>\n","metadata":{"papermill":{"duration":0.052259,"end_time":"2021-11-06T21:17:52.443218","exception":false,"start_time":"2021-11-06T21:17:52.390959","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">1.3 DATASET OVERVIEW</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">GENERAL INFORMATION</b>\n\nIn the [**previous HappyWhale competition**](https://www.kaggle.com/c/humpback-whale-identification), the task was to predict individual humpback whales from images of their flukes. Whales and dolphins in this dataset can be identified by shapes, features and markings (some natural, some acquired) of dorsal fins, backs, heads and flanks. Some species and some individuals have highly distinct features, others are very much less distinct. Further, individual features may change over time. \n\nThis competition expands that task significantly: \n* data in this competition contains images of over **15,000 unique individual marine mammals from 30 different species collected from 28 different research organizations**. \n* Individuals have been manually identified and given an **`individual_id`** by marine researches\n\nYour task is to correctly identify these individuals in the images. It's a challenging task that has the potential to drive significant advancements in understanding and protecting marine mammals across the globe.\n\nAn important note about data quality: \n* Bringing together this dataset from many different research organization posed a number of practical challenges. \n* Significant effort has been made to minimize data quality issues and as well as to minimize leakage as much as possible.\n* There are undoubtably issues. \n* We encourage the community to report these things so that future versions of the data can be improved, but unless there is a significant issue, we don't expect to make updates to the data during the competition.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">DISCOVERED  INFORMATION [TENTATIVE]</b>\n\nTBD","metadata":{"papermill":{"duration":0.052579,"end_time":"2021-11-06T21:17:52.548406","exception":false,"start_time":"2021-11-06T21:17:52.495827","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"background_information\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #3eb489; background-color: #ffffff;\" id=\"setup\">2&nbsp;&nbsp;SETUP&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>\n\n---\n","metadata":{"papermill":{"duration":0.052944,"end_time":"2021-11-06T21:17:52.65284","exception":false,"start_time":"2021-11-06T21:17:52.599896","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">2.1 ACCELERATOR DETECTION</h3>\n\n---\n\nIn order to use **`TPU`**, we use **`TPUClusterResolver`** for the initialization which is necessary to connect to the remote cluster and initialize cloud TPUs. Let's go over two important points\n\n1. When using TPU on Kaggle, you don't need to specify arguments for **`TPUClusterResolver`**\n2. However, on **G**oogle **C**ompute **E**ngine (**GCE**), you will need to do the following:\n\n<br>\n\n```python\n# The name you gave to the TPU to use\nTPU_WORKER = 'my-tpu-name'\n\n# or you can also specify the grpc path directly\n# TPU_WORKER = 'grpc://xxx.xxx.xxx.xxx:8470'\n\n# The zone you chose when you created the TPU to use on GCP.\nZONE = 'us-east1-b'\n\n# The name of the GCP project where you created the TPU to use on GCP.\nPROJECT = 'my-tpu-project'\n\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver(tpu=TPU_WORKER, zone=ZONE, project=PROJECT)\n```\n\n<div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">🛑 &nbsp; WARNING:</b><br><br>- Although the Tensorflow documentation says it is the <b>project name</b> that should be provided for the argument <b><code>`project`</code></b>, it is actually the <b>Project ID</b>, that you should provide. This can be found on the GCP project dashboard page.<br>\n</div>\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📖 &nbsp; REFERENCES:</b><br><br>\n    - <a href=\"https://www.tensorflow.org/guide/tpu#tpu_initialization\"><b>Guide - Use TPUs</b></a><br>\n    - <a href=\"https://www.tensorflow.org/api_docs/python/tf/distribute/cluster_resolver/TPUClusterResolver\"><b>Doc - TPUClusterResolver</b></a><br>\n\n</div>","metadata":{"papermill":{"duration":0.053821,"end_time":"2021-11-06T21:17:52.761303","exception":false,"start_time":"2021-11-06T21:17:52.707482","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(f\"\\n... ACCELERATOR SETUP STARTING ...\\n\")\n\n# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    TPU = tf.distribute.cluster_resolver.TPUClusterResolver()  \nexcept ValueError:\n    TPU = None\n\nif TPU:\n    print(f\"\\n... RUNNING ON TPU - {TPU.master()}...\")\n    tf.config.experimental_connect_to_cluster(TPU)\n    tf.tpu.experimental.initialize_tpu_system(TPU)\n    strategy = tf.distribute.experimental.TPUStrategy(TPU)\nelse:\n    print(f\"\\n... RUNNING ON CPU/GPU ...\")\n    # Yield the default distribution strategy in Tensorflow\n    #   --> Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy() \n\n# What Is a Replica?\n#    --> A single Cloud TPU device consists of FOUR chips, each of which has TWO TPU cores. \n#    --> Therefore, for efficient utilization of Cloud TPU, a program should make use of each of the EIGHT (4x2) cores. \n#    --> Each replica is essentially a copy of the training graph that is run on each core and \n#        trains a mini-batch containing 1/8th of the overall batch size\nN_REPLICAS = strategy.num_replicas_in_sync\n    \nprint(f\"... # OF REPLICAS: {N_REPLICAS} ...\\n\")\n\nprint(f\"\\n... ACCELERATOR SETUP COMPLTED ...\\n\")","metadata":{"papermill":{"duration":0.07574,"end_time":"2021-11-06T21:17:52.892074","exception":false,"start_time":"2021-11-06T21:17:52.816334","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-02-23T14:29:35.559433Z","iopub.execute_input":"2022-02-23T14:29:35.560290Z","iopub.status.idle":"2022-02-23T14:29:51.773932Z","shell.execute_reply.started":"2022-02-23T14:29:35.560224Z","shell.execute_reply":"2022-02-23T14:29:51.773240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">2.2 COMPETITION DATA ACCESS</h3>\n\n---\n\nTPUs read data must be read directly from **G**oogle **C**loud **S**torage **(GCS)**. Kaggle provides a utility library – **`KaggleDatasets`** – which has a utility function **`.get_gcs_path`** that will allow us to access the location of our input datasets within **GCS**.<br><br>\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📌 &nbsp; TIPS:</b><br><br>- If you have multiple datasets attached to the notebook, you should pass the name of a specific dataset to the <b><code>`get_gcs_path()`</code></b> function. <i>In our case, the name of the dataset is the name of the directory the dataset is mounted within.</i><br><br>\n</div>","metadata":{"papermill":{"duration":0.053551,"end_time":"2021-11-06T21:17:52.999073","exception":false,"start_time":"2021-11-06T21:17:52.945522","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"\\n... DATA ACCESS SETUP STARTED ...\\n\")\n\nif TPU:\n    # Google Cloud Dataset path to training and validation images\n    DATA_DIR = os.path.join(KaggleDatasets().get_gcs_path('happywhaletfrecords512x880'), \"512x880\")\n    save_locally = tf.saved_model.SaveOptions(experimental_io_device='/job:localhost')\n    with strategy.scope():\n        load_locally = tf.saved_model.LoadOptions(experimental_io_device='/job:localhost')\nelse:\n    # Local path to training and validation images\n    DATA_DIR = os.path.join('/kaggle/input/happywhaletfrecords512x880', \"512x880\")\n    save_locally = None\n    load_locally = None\n    \nprint(f\"\\n... DATA DIRECTORY PATH IS:\\n\\t--> {DATA_DIR}\")\n\nprint(f\"\\n... IMMEDIATE CONTENTS OF DATA DIRECTORY IS:\")\nfor file in tf.io.gfile.glob(os.path.join(DATA_DIR, \"*\")): print(f\"\\t--> {file}\")\n\nprint(\"\\n\\n... DATA ACCESS SETUP COMPLETED ...\\n\")","metadata":{"papermill":{"duration":0.0797,"end_time":"2021-11-06T21:17:53.133972","exception":false,"start_time":"2021-11-06T21:17:53.054272","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-02-23T14:29:51.775287Z","iopub.execute_input":"2022-02-23T14:29:51.775483Z","iopub.status.idle":"2022-02-23T14:29:52.267756Z","shell.execute_reply.started":"2022-02-23T14:29:51.775459Z","shell.execute_reply":"2022-02-23T14:29:52.266855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">2.3 LEVERAGING XLA OPTIMIZATIONS</h3>\n\n---\n\n\n**XLA** (Accelerated Linear Algebra) is a domain-specific compiler for linear algebra that can accelerate TensorFlow models with potentially no source code changes. **The results are improvements in speed and memory usage**.\n\n<br>\n\nWhen a TensorFlow program is run, all of the operations are executed individually by the TensorFlow executor. Each TensorFlow operation has a precompiled GPU/TPU kernel implementation that the executor dispatches to.\n\nXLA provides us with an alternative mode of running models: it compiles the TensorFlow graph into a sequence of computation kernels generated specifically for the given model. Because these kernels are unique to the model, they can exploit model-specific information for optimization.<br><br>\n\n<div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">🛑 &nbsp; WARNING:</b><br><br>- XLA can not currently compile functions where dimensions are not inferrable: that is, if it's not possible to infer the dimensions of all tensors without running the entire computation<br>\n</div>\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📌 &nbsp; NOTE:</b><br><br>- XLA compilation is only applied to code that is compiled into a graph (in <b>TF2</b> that's only a code inside <b><code>tf.function</code></b>).<br>- The <b><code>jit_compile</code></b> API has must-compile semantics, i.e. either the entire function is compiled with XLA, or an <b><code>errors.InvalidArgumentError</code></b> exception is thrown)\n</div>\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📖 &nbsp; REFERENCE:</b><br><br>    - <a href=\"https://www.tensorflow.org/xla\"><b>XLA: Optimizing Compiler for Machine Learning</b></a><br>\n</div>","metadata":{"papermill":{"duration":0.051494,"end_time":"2021-11-06T21:17:53.236017","exception":false,"start_time":"2021-11-06T21:17:53.184523","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(f\"\\n... XLA OPTIMIZATIONS STARTING ...\\n\")\n\nprint(f\"\\n... CONFIGURE JIT (JUST IN TIME) COMPILATION ...\\n\")\n# enable XLA optmizations (10% speedup when using @tf.function calls)\ntf.config.optimizer.set_jit(True)\n\nprint(f\"\\n... XLA OPTIMIZATIONS COMPLETED ...\\n\")","metadata":{"papermill":{"duration":0.128128,"end_time":"2021-11-06T21:17:53.442803","exception":false,"start_time":"2021-11-06T21:17:53.314675","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-02-23T14:29:52.270086Z","iopub.execute_input":"2022-02-23T14:29:52.270689Z","iopub.status.idle":"2022-02-23T14:29:52.276057Z","shell.execute_reply.started":"2022-02-23T14:29:52.270645Z","shell.execute_reply":"2022-02-23T14:29:52.275203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">2.4 BASIC DATA DEFINITIONS & INITIALIZATIONS</h3>\n\n---\n","metadata":{"papermill":{"duration":0.090507,"end_time":"2021-11-06T21:17:53.624612","exception":false,"start_time":"2021-11-06T21:17:53.534105","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"\\n... BASIC DATA SETUP STARTING ...\\n\\n\")\n\nK_FOLDS = N_FOLDS = 10 # More later...\nALL_TFRECORDS = tf.io.gfile.glob(os.path.join(DATA_DIR, \"**/*.tfrec\"))\ntfrecord_map = {\n    \"train\":{f\"fold_{i}\":[_tfrec for _tfrec in ALL_TFRECORDS if f\"train_fold_{i}\" in _tfrec] for i in range(1, N_FOLDS+1)},\n    \"test\":[_tfrec for _tfrec in ALL_TFRECORDS if \"test__\" in _tfrec]\n}\n\ntrain_df = pd.read_csv(\"../input/happy-whale-and-dolphin/train.csv\")\nFIX_NAME_MAPPING = {\"bottlenose_dolpin\":\"bottlenose_dolphin\", \n                    \"kiler_whale\":\"killer_whale\",\n                    \"pilot_whale\":\"short_finned_pilot_whale\",\n                    \"globis\":\"short_finned_pilot_whale\"}\ntrain_df[\"species\"] = train_df[\"species\"].apply(lambda x: x if x not in FIX_NAME_MAPPING.keys() else FIX_NAME_MAPPING[x])\n\nall_species = sorted(train_df.species.unique().tolist())\nspecies_int2str_lbl_map = {i:_s for i,_s in enumerate(all_species)}\nspecies_str2int_lbl_map = {v:k for k,v in species_int2str_lbl_map.items()}\nN_SPECIES = len(all_species)\n\nall_individuals = sorted(train_df.individual_id.unique().tolist())\nind_int2str_lbl_map = {i:_s for i,_s in enumerate(all_individuals)}\nind_str2int_lbl_map = {v:k for k,v in ind_int2str_lbl_map.items()}\nN_INDIVIDUALS = len(all_individuals)\n\ntrain_df[\"ind_sparse_lbl\"] = train_df[\"individual_id\"].map(ind_str2int_lbl_map)\ntrain_df[\"species_sparse_lbl\"] = train_df[\"species\"].map(species_str2int_lbl_map)\n\nss_df = pd.read_csv(\"../input/happy-whale-and-dolphin/sample_submission.csv\")\nss_df[\"img_path\"] = \"../input/happy-whale-and-dolphin/test_images/\"+ss_df.image\n\nsub_train_dfs = [pd.read_csv(f\"/kaggle/input/happywhaletfrecords512x880/train_df_fold_{i}.csv\") for i in range(1,K_FOLDS+1)]\nN_TEST = len(ss_df)\n\nprint(\"\\n\\n... BASIC DATA SETUP FINISHING ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-02-23T14:29:52.277292Z","iopub.execute_input":"2022-02-23T14:29:52.277565Z","iopub.status.idle":"2022-02-23T14:29:52.826013Z","shell.execute_reply.started":"2022-02-23T14:29:52.277533Z","shell.execute_reply":"2022-02-23T14:29:52.825217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cls_counts = {ind_str2int_lbl_map[k]:v for k,v in train_df.individual_id.value_counts().items()} \ncls_min_count = min(list(cls_counts.values()))\n\n# This will result in very small weighting for\n# 'common' individuals and a weighting of\n# 1.00 for the rarest individuals (probably only 1 image)\norig_cls_wt = {k:cls_min_count/v for k,v in cls_counts.items()}\n\n# One approach is to raise all the weights to a small value\n# between 0 and 1. \n# The closer the value is to 0 the closer the weights will be to 1. \n# The closer the value is to 1 the closer to the original values. \nflat_pwr = 0.1\nflat_cls_wt_1 = {k:(cls_min_count/v)**flat_pwr for k,v in cls_counts.items()}\n\n# Another step would be to use a slightly higher value for the flat_pwr\n# term, and clip the weights to a minimum to prevent certain very common\n# individuals from skewing the class balance\nflat_pwr = 0.25\nclip_val = 0.25\nflat_cls_wt_2 = {k:max(clip_val, (cls_min_count/v)**flat_pwr) for k,v in cls_counts.items()}\n\nfig = px.histogram(orig_cls_wt.values(), labels={\"value\": \"class weight\"}, title=\"<b>Original Weight Distribution</b>\", range_x=[0,1], log_y=True)\nfig.update_layout(showlegend=False) \nfig.show()\n\nfig = px.histogram(flat_cls_wt_1.values(), labels={\"value\": \"class weight\"}, title=\"<b>Shifted Weight Distribution</b>\", range_x=[0,1], log_y=True)\nfig.update_layout(showlegend=False) \nfig.show()\n\nfig = px.histogram(flat_cls_wt_2.values(), labels={\"value\": \"class weight\"}, title=\"<b>Shifted and Clipped Weight Distribution</b>\", range_x=[0,1], log_y=True)\nfig.update_layout(showlegend=False) \nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-02-23T14:29:52.827944Z","iopub.execute_input":"2022-02-23T14:29:52.828198Z","iopub.status.idle":"2022-02-23T14:29:53.776650Z","shell.execute_reply.started":"2022-02-23T14:29:52.828169Z","shell.execute_reply":"2022-02-23T14:29:53.776004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n\n<a id=\"helper_functions\"></a>\n\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #3eb489; background-color: #ffffff;\" id=\"helper_functions\">\n    3&nbsp;&nbsp;HELPER FUNCTION & CLASSES&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a>\n</h1>\n\n---","metadata":{"papermill":{"duration":0.054893,"end_time":"2021-11-06T21:17:54.695576","exception":false,"start_time":"2021-11-06T21:17:54.640683","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def flatten_l_o_l(nested_list):\n    \"\"\" Flatten a list of lists \"\"\"\n    return [item for sublist in nested_list for item in sublist]\n\n\ndef decode_image(image_data, resize_to=(512,880,3)):\n    \"\"\" Function to decode the tf.string containing image information \n    \n    \n    Args:\n        image_data (tf.string): String containing encoded image data from tf.Example\n        resize_to (tuple, optional): Size that we will reshape the tensor to (required for TPU)\n        to_norm_val (int, optional): The value that will allow us to normalize the tensor to 0-1\n    \n    Returns:\n        Tensor containing the resized single-channel image in the appropriate dtype\n    \"\"\"\n    image = tf.image.decode_png(image_data, channels=3, dtype=tf.uint8)\n    image = tf.reshape(image, resize_to)\n    return tf.cast(image, tf.float32)\n\n\ndef decode_example(serialized_example, is_test=False, return_ind_id=False, return_img_id=False, img_shape=(512,880,3)):\n    \"\"\" Parses a set of features and label from the given `serialized_example`.\n        \n        It is used as a map function for `dataset.map`\n\n    Args:\n        serialized_example (tf.Example): A serialized example containing the\n            following features:\n                – 'image'\n                – 'image_id'\n                – 'label'\n        is_test (bool, optional): Whether to allow for the label feature\n        \n    Returns:\n        A decoded tf.data.Dataset object representing the tfrecord dataset\n    \"\"\"\n    feature_dict = {\n        'image': tf.io.FixedLenFeature(shape=[], dtype=tf.string, default_value=''),\n    }\n    \n    if not is_test:\n        feature_dict[\"label\"] = tf.io.FixedLenFeature(shape=[], dtype=tf.int64, default_value=0)\n        feature_dict[\"ind_id\"] = tf.io.FixedLenFeature(shape=[], dtype=tf.int64, default_value=0)\n    feature_dict[\"image_id\"] = tf.io.FixedLenFeature(shape=[], dtype=tf.string, default_value='')\n    \n    # Define a parser\n    features = tf.io.parse_single_example(serialized_example, features=feature_dict)\n    \n    # Decode the tf.string\n    image = decode_image(features['image'], resize_to=img_shape)\n    \n    # Figure out the correct information to return\n    if is_test:\n        image_id = features[\"image_id\"] \n        return image, image_id\n    else:\n        label = features[\"label\"]\n        if (return_ind_id or return_img_id):\n            ind_id = features[\"ind_id\"]\n            image_id = features[\"image_id\"]\n            return image, label, ind_id, image_id\n        else:\n            return image, label\n    \nDEFAULT_AUG_CONFIG = dict(\n    random_hue=dict(max_delta=0.01,), random_brightness=dict(max_delta=0.1,),\n    random_saturation=dict(lower=0.75, upper=1.25), random_contrast=dict(lower=0.875, upper=1.125)\n)\ndef augment(img_batch, cutout_central_fraction=0.95, cutout_ratio=26, aug_config=DEFAULT_AUG_CONFIG):\n    \n    # SEEDING & KERNEL INIT\n    img_batch = tf.cast(img_batch, tf.float32)\n    \n    # Random Flip\n    img_batch = tf.image.random_flip_left_right(img_batch,)\n    \n    # Central Crop\n    K = tf.random.uniform((1,), minval=0, maxval=4, dtype=tf.dtypes.int32)[0]    \n    img_batch = tf.cond(K==1, lambda: tf.image.resize(\n        tf.image.central_crop(img_batch, central_fraction=cutout_central_fraction), \n        (IMG_SHAPE[0], IMG_SHAPE[1])\n    ), lambda: img_batch)\n    \n    # Random Cutout (black squares)\n    img_batch = tfa.image.random_cutout(img_batch, mask_size=(2*(IMG_SHAPE[0]//cutout_ratio),2*(IMG_SHAPE[0]//cutout_ratio)), constant_values=0)\n    \n    # Visual Augmentations\n    img_batch = tf.image.random_hue(img_batch, **aug_config[\"random_hue\"])\n    img_batch = tf.image.random_saturation(img_batch, **aug_config[\"random_saturation\"])\n    img_batch = tf.image.random_contrast(img_batch, **aug_config[\"random_contrast\"])\n    img_batch = tf.image.random_brightness(img_batch, **aug_config[\"random_brightness\"])\n    \n    return tf.cast(img_batch, tf.float32)\n\n\n# UNNECESSARY NOW\n# # build a lookup table so we can do a dictionary lookup in graph mode\n# s2i_table = tf.lookup.StaticHashTable(\n#     initializer=tf.lookup.KeyValueTensorInitializer(\n#         keys=tf.constant(list(ind_str2int_lbl_map.keys())),\n#         values=tf.constant(list(ind_str2int_lbl_map.values())),\n#     ),\n#     default_value=tf.constant(-1),\n#     name=\"str2int_mapping\"\n# )","metadata":{"papermill":{"duration":0.098071,"end_time":"2021-11-06T21:17:54.848036","exception":false,"start_time":"2021-11-06T21:17:54.749965","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-02-23T14:29:53.778008Z","iopub.execute_input":"2022-02-23T14:29:53.778571Z","iopub.status.idle":"2022-02-23T14:29:53.799817Z","shell.execute_reply.started":"2022-02-23T14:29:53.778536Z","shell.execute_reply":"2022-02-23T14:29:53.798768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Augmentation Example\n\nIMG_SHAPE = (512,880,3)\n\nex_img_1 = tf.expand_dims(cv2.resize(cv2.imread(\"../input/happy-whale-and-dolphin/train_images/00167e8375c967.jpg\")[..., ::-1], (880,512)), axis=0)\nex_img_2 = tf.expand_dims(cv2.resize(cv2.imread(\"../input/happy-whale-and-dolphin/train_images/0028f6fa123686.jpg\")[..., ::-1], (880,512)), axis=0)\nex_img_3 = tf.expand_dims(cv2.resize(cv2.imread(\"../input/happy-whale-and-dolphin/train_images/0007c33415ce37.jpg\")[..., ::-1], (880,512)), axis=0)\nex_batch = tf.cast(tf.concat([ex_img_1, ex_img_2, ex_img_3], axis=0), tf.float32)\naug_batch = augment(ex_batch)\n\nplt.figure(figsize=(20, 15))\nfor i in range(3):\n    plt.subplot(3,2,i*2+1)\n    plt.imshow(ex_batch[i]/255.)\n    plt.title(f\"Original Image {i+1}\", fontweight=\"bold\")\n    \n    plt.subplot(3,2,i*2+2)\n    plt.imshow(aug_batch[i]/255.)\n    plt.title(f\"Augmented Image {i+1}\", fontweight=\"bold\")\n    \nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-23T14:29:53.800820Z","iopub.execute_input":"2022-02-23T14:29:53.801762Z","iopub.status.idle":"2022-02-23T14:29:57.090717Z","shell.execute_reply.started":"2022-02-23T14:29:53.801731Z","shell.execute_reply":"2022-02-23T14:29:57.089455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"KAGGLE REFERENCE  \n        --> https://www.kaggle.com/ragnar123/unsupervised-baseline-arcface","metadata":{}},{"cell_type":"code","source":"class ArcMarginProduct(tf.keras.layers.Layer):\n    \"\"\"\n    KAGGLE REFERENCE  \n        --> https://www.kaggle.com/ragnar123/unsupervised-baseline-arcface\n        \n    Implements large margin arc distance.\n\n    Reference:\n        https://arxiv.org/pdf/1801.07698.pdf\n        https://github.com/lyakaap/Landmark2019-1st-and-3rd-Place-Solution/\n            blob/master/src/modeling/metric_learning.py\n    \"\"\"\n    def __init__(self, n_classes, s=30, m=0.3,\n                 easy_margin=False, ls_eps=0.0,  **kwargs):\n        super(ArcMarginProduct, self).__init__(**kwargs)\n\n        self.n_classes = n_classes\n        self.s, self.m = s, m\n        self.ls_eps = ls_eps\n        self.easy_margin = easy_margin\n        self.sin_m, self.cos_m = tf.math.sin(m), tf.math.cos(m)\n        self.mm, self.th = tf.math.sin(math.pi-m)*m, tf.math.cos(math.pi-m)\n\n    def get_config(self):\n        config = super().get_config().copy()\n        config.update({\n            'n_classes': self.n_classes, 's': self.s, 'm': self.m,\n            'ls_eps': self.ls_eps, 'easy_margin': self.easy_margin,\n        })\n        return config\n\n    def build(self, input_shape):\n        super(ArcMarginProduct, self).build(input_shape[0])\n\n        self.W = self.add_weight(\n            name='W', shape=(int(input_shape[0][-1]), self.n_classes),\n            initializer='glorot_uniform', dtype='float32', \n            trainable=True, regularizer=None)\n\n    def call(self, inputs):\n        X, y = inputs\n        y = tf.cast(y, dtype=tf.int32)\n        cosine = tf.matmul(\n            tf.math.l2_normalize(X, axis=1),\n            tf.math.l2_normalize(self.W, axis=0)\n        )\n        sine = tf.math.sqrt(1.0 - tf.math.pow(cosine, 2))\n        phi = cosine * self.cos_m - sine * self.sin_m\n        \n        if self.easy_margin:\n            phi = tf.where(cosine > 0, phi, cosine)\n        else:\n            phi = tf.where(cosine > self.th, phi, cosine - self.mm)\n        \n        one_hot = tf.cast(\n            tf.one_hot(y, depth=self.n_classes),\n            dtype=cosine.dtype\n        )\n        \n        if self.ls_eps > 0:\n            one_hot = (1 - self.ls_eps) * one_hot + self.ls_eps / self.n_classes\n\n        output = (one_hot * phi) + ((1.0 - one_hot) * cosine)\n        output *= self.s\n        \n        return output","metadata":{"execution":{"iopub.status.busy":"2022-02-23T14:29:57.092652Z","iopub.execute_input":"2022-02-23T14:29:57.093132Z","iopub.status.idle":"2022-02-23T14:29:57.116620Z","shell.execute_reply.started":"2022-02-23T14:29:57.093077Z","shell.execute_reply":"2022-02-23T14:29:57.115721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n\n<a id=\"dataset_preparation\"></a>\n\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #3eb489; background-color: #ffffff;\" id=\"dataset_preparation\">4&nbsp;&nbsp;PREPARE THE DATASET&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>\n\nIn this section we prepare the **`tf.data.Datasets`** we will use for training and validation","metadata":{"papermill":{"duration":0.056443,"end_time":"2021-11-06T21:17:54.960518","exception":false,"start_time":"2021-11-06T21:17:54.904075","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">4.1 READ TFRECORD FILES - CREATE THE RAW DATASET(S)</h3>\n\n---\n\nHere we will leverage **`tf.data.TFRecordDataset`** to read the TFRecord files.\n* The simplest way is to specify a list of filenames (paths) of TFRecord files.\n* It is a subclass of **`tf.data.Dataset`**.\n\nThis newly created raw dataset contains **`tf.train.Example`** messages, and when iterated over it, we get scalar string tensors.","metadata":{"papermill":{"duration":0.053158,"end_time":"2021-11-06T21:17:55.070372","exception":false,"start_time":"2021-11-06T21:17:55.017214","status":"completed"},"tags":[]}},{"cell_type":"code","source":"tfds_map = {\"train\":{}, \"test\":[]}\ntfds_map[\"train\"] = {\n    k:tf.data.TFRecordDataset(v, num_parallel_reads=tf.data.AUTOTUNE) \\\n    for k,v in tfrecord_map[\"train\"].items()\n}\ntfds_map[\"test\"] = tf.data.TFRecordDataset(tfrecord_map[\"test\"], num_parallel_reads=tf.data.AUTOTUNE)\nprint(tfds_map)","metadata":{"papermill":{"duration":0.313817,"end_time":"2021-11-06T21:17:55.439001","exception":false,"start_time":"2021-11-06T21:17:55.125184","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-02-23T14:29:57.119319Z","iopub.execute_input":"2022-02-23T14:29:57.119724Z","iopub.status.idle":"2022-02-23T14:29:57.207512Z","shell.execute_reply.started":"2022-02-23T14:29:57.119678Z","shell.execute_reply":"2022-02-23T14:29:57.206627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">4.2 WHAT IF YOU DON'T KNOW THE FEATURES OF THE DATASET?</h3>\n\n---\n\nIf you are the author who created the TFRecord files, you definitely know how to define the feature description to parse the raw dataset.\n\nOtherwise, you can use like\n\n```python\nexample = tf.train.Example()\nexample.ParseFromString(serialized_example.numpy())\n```\n\nto check the information. You will get something like\n\n```python\nfeatures {\n    feature {\n        key: \"class\"\n        value {\n            int64_list {\n                value: 57\n            }\n        }\n    }\n    feature {\n        key: \"id\"\n        value {\n            bytes_list {\n                value: \"338ab7bac\"\n            }\n        }\n    }\n    feature {\n        key: \"image\"\n        value {\n            bytes_list {\n                value: ...\n            }\n        }\n    }\n    ...\n}\n```\n\nThis should give you enough information to define the feature description.","metadata":{"papermill":{"duration":0.05563,"end_time":"2021-11-06T21:17:55.55248","exception":false,"start_time":"2021-11-06T21:17:55.49685","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"\\n... RAW TFRECORD INVESTIGATION TO DETERMINE FEATURE DESCRIPTIONS STARTED ...\\n\")\n\nprint(\"\\n... EXAMPLE OF TRUNCATED RAW TFRECORD/TFEXAMPLE FROM TRAINING DATASET TO SHOW HOW TO FIND FEATURE DESCRIPTIONS:\\n\")\n# See an example\nfor raw in tfds_map[\"train\"][\"fold_1\"].take(1):\n    example = tf.train.Example()\n    example.ParseFromString(raw.numpy())\n    for i, (k,v) in enumerate(example.features.feature.items()):\n        print(f\"\\tFEATURE #{i+1}\")\n        print(f\"\\t\\t--> KEY = {k}\")\n        if k!=\"image\":\n            try:\n                print(f\"\\t\\t\\t--> TRUNCATED-VALUE = {v.bytes_list.value[0][:25]} ...\\n\")\n            except:\n                print(f\"\\t\\t\\t--> TRUNCATED-VALUE = {v.int64_list.value[:15]} ...\\n\")\n        else:\n            print(f\"\\t\\t\\t--> TRUNCATED-VALUE = {str(v.bytes_list.value[0][:25])} ...\\n\")         \n\nprint(\"\\n... RAW TFRECORD INVESTIGATION TO DETERMINE FEATURE DESCRIPTIONS COMPLETED ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-02-23T14:29:57.208599Z","iopub.execute_input":"2022-02-23T14:29:57.208829Z","iopub.status.idle":"2022-02-23T14:29:57.535363Z","shell.execute_reply.started":"2022-02-23T14:29:57.208804Z","shell.execute_reply":"2022-02-23T14:29:57.533289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">4.3 PARSE THE RAW DATASET(S)</h3>\n\n---\n\n\nThe general recipe to parse the string tensors in the raw dataset looks something like this:\n\n<br>\n\n**STEP 1.**  Create a description of the features. For example:\n\n```python\nfeature_description = {    \n    'feature0': tf.io.FixedLenFeature([], tf.int64),\n    'feature1': tf.io.FixedLenFeature([], tf.string),\n    'feature2': tf.io.FixedLenFeature([], tf.float32),\n    ...\n}\n```\n\n<br>\n\n**STEP 2.**  Define a parsing function by using `tf.io.parse_single_example` and the defined feature description.\n```python\ndef _parse_function(example):\n    \"\"\"\n    Args:\n        example: A string tensor representing a `tf.train.Example`.\n    \"\"\"\n\n    # Parse `example`.\n    parsed_example = tf.io.parse_single_example(example, feature_description)\n\n    return parsed_example\n```\n\n<br>\n\n**STEP 3.**  Map the raw dataset by `_parse_function`.\n```python\ndataset = raw_dataset.map(_parse_function)\n```\n\n<br>\n\n---\n\n<br>\n\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📌 &nbsp; NOTE:</b><br><br>- The parsed images are <code><b>`tf.string`</b></code>, which are then decoded with <code><b>`tf.image.decode_png`</b></code> which is an alias for <code><b>`tf.io.decode_png`</b></code><br>- The InChI strings and Image IDs will just be left as byte string tensors.\n</div>\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📖 &nbsp; REFERENCE:</b><br><br>\n    - <a href=\"https://www.tensorflow.org/tutorials/load_data/tfrecord\"><b>Tutorial - TFRecord and tf.Example</b></a><br>\n    - <a href=\"https://www.tensorflow.org/api_docs/python/tf/data/TFRecordDataset\"><b>TFRecordDataset Documentation</b></a><br>\n    - <a href=\"https://www.tensorflow.org/api_docs/python/tf/io/decode_png\"><b>Decoding PNGs Documentation</b></a><br>\n</div>\n","metadata":{"papermill":{"duration":0.684105,"end_time":"2021-11-06T21:18:30.682923","exception":false,"start_time":"2021-11-06T21:18:29.998818","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"\\n... DECODING RAW TFRECORD DATASETS STARTING ...\\n\")\n\ntfds_map[\"train\"] = {\n    k:v.map(lambda x: decode_example(x, is_test=False, return_img_id=True), \n            num_parallel_calls=tf.data.AUTOTUNE) for k,v in tfds_map[\"train\"].items()\n}\ntfds_map[\"test\"]=tfds_map[\"test\"].map(lambda x: decode_example(x, is_test=True), num_parallel_calls=tf.data.AUTOTUNE)\nprint(tfds_map)\n\nprint(f\"\\n... THE DECODED TF.DATA.TFRECORDDATASET OBJECT:\" \\\n      f\"\\n\\t--> ((image), (image_id - optional), (ind_id - optional), (label))\" \\\n      f\"\\n\\t--> {tfds_map['train'][f'fold_1']}\\n\")\n\nprint(\"\\n... 4 EXAMPLES OF TRAIN IMAGES AND LABELS AFTER DECODING AND AUGMENTATION ...\")\nfor i, (img, label, ind, img_id) in enumerate(tfds_map[\"train\"][f\"fold_1\"].take(2)):\n    display(train_df[train_df.image==f\"{img_id.numpy().decode()}.jpg\"])\n    print(f\"\\nIMAGE SHAPE : {img.shape}\")\n    print(f\"SPECIES LABEL : {species_int2str_lbl_map[label.numpy()]}\")\n    print(f\"INDIVIDUAL LABEL: {ind_int2str_lbl_map[ind.numpy()]}\")\n    print(f\"IMAGE ID: {img_id.numpy().decode()}\")\n    plt.figure(figsize=(10,10))\n    plt.imshow(img.numpy()/255., cmap=\"gray\")\n    plt.axis(False)\n    plt.tight_layout()\n    plt.show()\n    \nprint(\"\\n... 4 EXAMPLES OF TEST IMAGES AND IMAGE IDS AFTER DECODING ...\")\nfor i, (img, img_id) in enumerate(tfds_map[\"test\"].take(2)):\n    print(f\"\\nIMAGE SHAPE : {img.shape}\")\n    print(f\"IMAGE ID: {img_id.numpy().decode()}\")\n    plt.figure(figsize=(10,10))\n    plt.imshow(img.numpy()/255., cmap=\"gray\")\n    plt.axis(False)\n    plt.tight_layout()\n    plt.show()\n\nprint(\"\\n... DECODING RAW TFRECORD DATASETS COMPLETED ...\\n\")\n","metadata":{"execution":{"iopub.status.busy":"2022-02-23T14:29:57.537256Z","iopub.execute_input":"2022-02-23T14:29:57.537577Z","iopub.status.idle":"2022-02-23T14:30:00.028331Z","shell.execute_reply.started":"2022-02-23T14:29:57.537524Z","shell.execute_reply":"2022-02-23T14:30:00.027463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">4.4 ADD PIPELINE FUNCTIONALITY</h3>\n\n---","metadata":{}},{"cell_type":"code","source":"# Model Training Parameters\nINPUT_SHAPE = (512,880,3)\nREPLICA_BATCH_SIZE = 16\nBATCH_SIZE = REPLICA_BATCH_SIZE*N_REPLICAS\nBUFFER_SIZE=8*BATCH_SIZE\nN_EPOCHS=40\n\n# MODEL_NAME = \"efficientnet_v2_imagenet21k_m\"\nMODEL_NAME = \"efficientnet_b7_noisy\"\nTFHUB_PATH = f\"https://tfhub.dev/google/imagenet/{MODEL_NAME}/feature_vector/2\"\nMODEL_PATH = \"\"\nN_DENSE = 768\nDROPOUT = 0.25\nUSE_GEM=True\n\n# Learning Rate Stuff\nINIT_LR = 0.00001\nMAX_LR  = 0.00111","metadata":{"execution":{"iopub.status.busy":"2022-02-23T14:36:16.885848Z","iopub.execute_input":"2022-02-23T14:36:16.886213Z","iopub.status.idle":"2022-02-23T14:36:16.893702Z","shell.execute_reply.started":"2022-02-23T14:36:16.886177Z","shell.execute_reply":"2022-02-23T14:36:16.892519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">5.1 TRAINING</h3>\n\n---\n","metadata":{}},{"cell_type":"code","source":"def get_train_val_split(ds_map, val_fold=\"1\"):\n    # Get shuffled training folds as tf.data.Dataset\n    train_folds = [\n        str(x) for x \\\n        in random.sample(range(1, 11), 10) \\\n        if str(x)!=str(val_fold)\n    ]\n    \n    print(train_folds)\n    print(val_fold)\n    _train_ds = ds_map[\"train\"][f\"fold_{train_folds[0]}\"]\n    for _fold_num in train_folds[1:]:\n        _train_ds = _train_ds.concatenate(ds_map[\"train\"][f\"fold_{_fold_num}\"])\n        \n    # Get validation dataset as tf.data.Dataset\n    _val_ds = ds_map[\"train\"][f\"fold_{val_fold}\"]\n    return _train_ds, _val_ds\n\ndef return_padded_ds(_ds, ds_length, is_test=False):\n    def get_n_pad(_ds_length):\n        return BATCH_SIZE-_ds_length%BATCH_SIZE\n    # Get amount to pad\n    n_pad = get_n_pad(ds_length)\n    \n    if is_test:\n        _pad_ds = tf.data.Dataset.from_tensor_slices((\n                    tf.zeros((n_pad, *INPUT_SHAPE), dtype=tf.float32),    # Fake Images\n                    tf.constant([\"000000000000\",]*n_pad, dtype=tf.string) # Fake Image ID\n        ))\n        _ds = _ds.concatenate(_pad_ds) # update in place\n    else:\n        _pad_ds = tf.data.Dataset.from_tensor_slices((\n                tf.zeros((n_pad, *INPUT_SHAPE), dtype=tf.float32),    # Fake Images\n                tf.zeros((n_pad,), dtype=tf.int64),                   # Fake Label\n                tf.zeros((n_pad,), dtype=tf.int64),                   # Fake Individual\n                tf.constant([\"000000000000\",]*n_pad, dtype=tf.string) # Fake Image ID\n        ))\n        _ds = _ds.concatenate(_pad_ds) # make new dataset\n    return _ds\n\ndef get_lr_callback(do_plot=True):\n    \"\"\"\n    https://www.kaggle.com/ragnar123/shopee-efficientnetb3-arcmarginproduct\n    \"\"\"\n    lr_start   = 0.000001\n    lr_max     = 0.000005*BATCH_SIZE\n    lr_min     = 0.000001\n    lr_ramp_ep = 5\n    lr_sus_ep  = 2\n    lr_decay   = 0.9\n   \n    def lrfn(epoch):\n        if epoch < lr_ramp_ep:\n            lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start   \n        elif epoch < lr_ramp_ep + lr_sus_ep:\n            lr = lr_max    \n        else:\n            lr = (lr_max - lr_min) * lr_decay**(epoch - lr_ramp_ep - lr_sus_ep) + lr_min    \n        return lr\n    \n    lr_callback = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose = True)\n    \n    fig = px.scatter(x=[x for x in range(N_EPOCHS)], y=[lrfn(x) for x in range(N_EPOCHS)], title=\"Learning Rate Schedule\", labels={\"x\":\"<b>Epoch Number</b>\", \"y\":\"<b>Learning Rate</b>\"})\n    fig.show()\n    return lr_callback\n\ndef plot_history(history, fold_num=\"1\"):\n    fig = px.line(history.history, \n                  x=range(len(history.history[\"loss\"])), \n                  y=[\"loss\", \"val_loss\"],\n                  labels={\"value\":\"Loss (log-axis)\", \"x\":\"Epoch #\"},\n                  title=f\"<b>FOLD {fold_num} MODEL - LOSS</b>\", log_y=True\n                  )\n    fig.show()\n\n    fig = px.line(history.history, \n                  x=range(len(history.history[\"acc\"])), \n                  y=[\"acc\", \"val_acc\"],\n                  labels={\"value\":\"Accuracy (log-axis)\", \"x\":\"Epoch #\"},\n                  title=f\"<b>FOLD {fold_num} MODEL - ACCURACY</b>\", log_y=True)\n    fig.show()\n\nclass GeMPoolingLayer(tf.keras.layers.Layer):\n    def __init__(self, p=3., train_p=False):\n        super().__init__()\n        if train_p:\n            self.p = tf.Variable(p, dtype=tf.float32)\n        else:\n            self.p = p\n        self.eps = 1e-6\n\n    def call(self, inputs: tf.Tensor, **kwargs):\n        inputs = tf.clip_by_value(inputs, clip_value_min=1e-6, clip_value_max=tf.reduce_max(inputs))\n        inputs = tf.pow(inputs, self.p)\n        inputs = tf.reduce_mean(inputs, axis=[1, 2], keepdims=False)\n        inputs = tf.pow(inputs, 1./self.p)\n        return inputs\n    \ndef get_model(backbone, input_shape=INPUT_SHAPE, use_GeM=False, freeze_bn=False, n_dense=512, dropout=0.025, n_classes=N_INDIVIDUALS):\n    \"\"\" TBD \"\"\"\n    # Define the ArcMargin head layer\n    MarginLayer = ArcMarginProduct(\n        n_classes=n_classes, name='head/arc_margin', dtype='float32',\n    )\n    GeMLayer = GeMPoolingLayer()\n    \n    # Define inputs\n    #   input_1 --> image\n    #   input_2 --> label\n    _input_1 = tf.keras.layers.Input(shape = INPUT_SHAPE, name = 'input_1')\n    _input_2 = tf.keras.layers.Input(shape = (),          name = 'input_2')\n    \n    x = backbone(_input_1)\n    \n    if use_GeM:\n        x = GeMLayer(x)\n    \n    x = tf.keras.layers.Dropout(dropout)(x)\n    x = tf.keras.layers.Dense(n_dense, name=\"embedding_out\")(x) # no activation\n    x = MarginLayer([x, _input_2])\n    _output = tf.keras.layers.Softmax(dtype='float32')(x)    \n    return tf.keras.models.Model(inputs = [_input_1, _input_2], outputs = [_output])\n\ndef get_embed_model(model, bb_layer_name=\"embedding_out\"):\n    _inputs = model.input[0]\n    _outputs = model.get_layer(bb_layer_name).output\n    return tf.keras.Model(inputs=_inputs, outputs=_outputs)\n\ndef get_embedding(_ds, _emodel, clip_pad_at, style=\"train\", to_disk=True, fold_num=\"1\"):\n    print(f\"\\n\\n\\n... GENERATING {style.upper()} EMBEDDINGS ...\\n\\n\")\n    _embeds, _labels, _individual_ids, _image_ids = [], [], [], []\n    \n    if style==\"test\":\n        for img_batch, img_id_batch in tqdm(_ds, total=clip_pad_at//BATCH_SIZE):\n            pred_batch = _emodel.predict(img_batch)\n            _embeds.append(pred_batch)\n            _image_ids.append(img_id_batch)    \n        _embeds = tf.concat(_embeds, axis=0).numpy()[:clip_pad_at]\n        _image_ids = tf.concat(_image_ids, axis=0).numpy()[:clip_pad_at]\n    else:\n        for img_batch, lbl_batch, ind_id_batch, img_id_batch in tqdm(_ds, total=clip_pad_at//BATCH_SIZE):\n            pred_batch = _emodel.predict(img_batch)\n            _embeds.append(pred_batch)\n            _image_ids.append(img_id_batch)    \n            _labels.append(lbl_batch)\n            _individual_ids.append(ind_id_batch)\n        _embeds = tf.concat(_embeds, axis=0).numpy()[:clip_pad_at]\n        _image_ids = tf.concat(_image_ids, axis=0).numpy()[:clip_pad_at]\n        _individual_ids = tf.concat(_individual_ids, axis=0).numpy()[:clip_pad_at]\n        _labels = tf.concat(_labels, axis=0).numpy()[:clip_pad_at]\n    \n    if to_disk:\n        np.save(f\"fold_{fold_num}__{style}_image_embeds\", _embeds)\n        np.save(f\"fold_{fold_num}__{style}_image_ids\", _image_ids)\n        if style!=\"test\":\n            np.save(f\"fold_{fold_num}__{style}_individual_ids\", _individual_ids)\n            np.save(f\"fold_{fold_num}__{style}_image_labels\", _labels)\n    else:\n        raise NotImplementedError()","metadata":{"execution":{"iopub.status.busy":"2022-02-23T14:30:00.037132Z","iopub.execute_input":"2022-02-23T14:30:00.037372Z","iopub.status.idle":"2022-02-23T14:30:00.077532Z","shell.execute_reply.started":"2022-02-23T14:30:00.037348Z","shell.execute_reply":"2022-02-23T14:30:00.076710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-02-23T14:38:03.731645Z","iopub.execute_input":"2022-02-23T14:38:03.731939Z","iopub.status.idle":"2022-02-23T14:38:03.738686Z","shell.execute_reply.started":"2022-02-23T14:38:03.731907Z","shell.execute_reply":"2022-02-23T14:38:03.737789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 0: Initialize Histories\nhistories = []\n\n# Step 1: Add Padding to Test Dataset\ntest_ds = return_padded_ds(tfds_map[\"test\"], ds_length=len(ss_df), is_test=True)\n\n# Step 2: Preprocess Test Dataset\ntest_ds = test_ds.batch(BATCH_SIZE, drop_remainder=True) \\\n                 .map(lambda x,y: (x/255., y)) \\\n                 .prefetch(tf.data.AUTOTUNE)\n\n# Step 3: Start Looping\nfor _k in range(7, K_FOLDS+1):\n    print(f\"\\n\\n\\n\\n\\n... STARTING TRAINING FOR FOLD {_k} ...\\n\\n\\n\")\n    # Step 4: Get train and val split based on fold\n    #            --> Train = 9 folds\n    #            --> Val   = 1 fold\n    train_ds, val_ds = get_train_val_split(tfds_map, val_fold=str(_k))\n    \n    # Step 5: Get a padded version of the datasets\n    N_TRAIN = sum([len(x) for i, x in enumerate(sub_train_dfs) if i!=(_k-1)])\n    N_VAL = len(sub_train_dfs[_k-1])\n    padded_train_ds = return_padded_ds(train_ds, ds_length=N_TRAIN)\n    padded_val_ds   = return_padded_ds(val_ds, ds_length=N_VAL)\n    \n    # Step 6: Preprocess the datasets\n    train_ds = train_ds.shuffle(BUFFER_SIZE) \\\n                       .batch(BATCH_SIZE, drop_remainder=True) \\\n                       .map(lambda x,y,z,a: (augment(x)/255., tf.cast(y, tf.uint8), tf.cast(z, tf.int32), a), num_parallel_calls=tf.data.AUTOTUNE) \\\n                       .prefetch(tf.data.AUTOTUNE)\n    \n    # No augmentation for validation or padded datasets\n    # We will still shuffle the validation dataset during training \n    # so we can see performance on all examples (as some will be dropped each epoch)\n    val_ds = val_ds.shuffle(BUFFER_SIZE) \\\n                   .batch(BATCH_SIZE, drop_remainder=True) \\\n                   .map(lambda x,y,z,a: (x/255., tf.cast(y, tf.uint8), tf.cast(z, tf.int32), a), num_parallel_calls=tf.data.AUTOTUNE) \\\n                   .prefetch(tf.data.AUTOTUNE)\n    padded_train_ds = padded_train_ds.batch(BATCH_SIZE, drop_remainder=True) \\\n                                     .map(lambda x,y,z,a: (x/255., tf.cast(y, tf.uint8), tf.cast(z, tf.int32), a), num_parallel_calls=tf.data.AUTOTUNE) \\\n                                     .prefetch(tf.data.AUTOTUNE)\n    padded_val_ds = padded_val_ds.batch(BATCH_SIZE, drop_remainder=True) \\\n                                 .map(lambda x,y,z,a: (x/255., tf.cast(y, tf.uint8), tf.cast(z, tf.int32), a), num_parallel_calls=tf.data.AUTOTUNE) \\\n                                 .prefetch(tf.data.AUTOTUNE)\n\n    # Step 7: Load the model within the strategy, compile and train\n    with strategy.scope():\n        if \"v2\" in MODEL_NAME:\n            model = get_model(tfhub.KerasLayer(TFHUB_PATH, trainable=True, load_options=load_locally), n_dense=N_DENSE, dropout=DROPOUT, n_classes=N_INDIVIDUALS, use_GeM=False)\n        else:\n            model = get_model(efn.EfficientNetB6(include_top=False, weights='noisy-student'), n_dense=N_DENSE, dropout=DROPOUT, n_classes=N_INDIVIDUALS, use_GeM=USE_GEM)\n        if _k==1:\n            print(model.summary())\n        \n        # ######################################################################\n        # Learning Rate Stuff\n        # ######################################################################\n        steps_per_epoch = N_TRAIN//BATCH_SIZE\n        clr = tfa.optimizers.CyclicalLearningRate(initial_learning_rate=INIT_LR,\n                                          maximal_learning_rate=MAX_LR,\n                                          scale_fn=lambda x: 1/(2.5**(x-1)),\n                                          step_size=4*steps_per_epoch)\n        step = np.arange(0, N_EPOCHS*steps_per_epoch)\n        lr = clr(step)\n        fig = px.scatter(x=step, y=list(lr.numpy()), \n                         title=\"<b>Cyclical Learning Rate Schedule</b>\",\n                         labels={\"x\":\"<b>Step</b>\", \"y\":\"<b>Learning Rate</b>\"})\n        fig.show()\n        # ######################################################################\n        \n        model.compile(optimizer=tf.keras.optimizers.Adam(clr),\n                     loss=tf.keras.losses.SparseCategoricalCrossentropy(),\n                     metrics=[\"acc\", tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5)])\n        \n        ckpt_cb = tf.keras.callbacks.ModelCheckpoint(f'fold_{_k}__{MODEL_NAME}__{IMG_SHAPE[0]}x{IMG_SHAPE[1]}', \n                                                     monitor='val_loss', save_best_only=True,\n                                                     options=save_locally, mode='min')\n        # lr_cb = get_lr_callback()\n        \n    # Step 8: Keep the history and plot - ADDED CLASS WEIGHTING\n    history = model.fit(train_ds.map(lambda a,b,c,d: ((a,c), c)), \n                        validation_data=val_ds.map(lambda a,b,c,d: ((a,c), c)), \n                        epochs=N_EPOCHS, callbacks = [ckpt_cb,],\n                        class_weight=flat_cls_wt_2)\n    plot_history(history)\n    histories.append(history)\n    \n    # Step 9: Create the embedding model\n    with strategy.scope():\n        emodel = get_embed_model(model)\n        emodel.trainable=False\n    \n    # Step 10: Generate and save embeddings\n    get_embedding(padded_train_ds, emodel, clip_pad_at=N_TRAIN, style=\"train\", fold_num=str(_k))\n    get_embedding(padded_val_ds, emodel, clip_pad_at=N_VAL, style=\"val\", fold_num=str(_k))\n    get_embedding(test_ds, emodel, clip_pad_at=N_TEST, style=\"test\", fold_num=str(_k))","metadata":{"execution":{"iopub.status.busy":"2022-02-23T17:08:09.107912Z","iopub.execute_input":"2022-02-23T17:08:09.108378Z","iopub.status.idle":"2022-02-23T17:08:13.863818Z","shell.execute_reply.started":"2022-02-23T17:08:09.108338Z","shell.execute_reply":"2022-02-23T17:08:13.862179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}