{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<br>\n\n<h2 style=\"text-align: center; font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: underline; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">Happy Whale and Dolphin <span style=\"font-size: 16px;\">(😊🐳&🐬)</span> - Create TFRecords</h2>\n<h5 style=\"text-align: center; font-family: Verdana; font-size: 12px; font-style: normal; font-weight: bold; text-decoration: None; text-transform: none; letter-spacing: 1px; color: black; background-color: #ffffff;\">CREATED BY: DARIEN SCHETTLER</h5>\n\n<br>\n\n---\n\n<br>\n\n<center><div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">🛑 &nbsp; WARNING:</b><br><br><b>THIS IS A WORK IN PROGRESS</b><br>\n</div></center>\n\n\n<center><div class=\"alert alert-block alert-warning\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 18px;\">👏 &nbsp; IF YOU FORK THIS OR FIND THIS HELPFUL &nbsp; 👏</b><br><br><b style=\"font-size: 22px; color: darkorange\">PLEASE UPVOTE!</b><br><br>This was a lot of work for me and while it may seem silly, it makes me feel appreciated when others like my work. 😅\n</div></center>\n\n\n","metadata":{"papermill":{"duration":0.120561,"end_time":"2021-11-06T21:15:09.611563","exception":false,"start_time":"2021-11-06T21:15:09.491002","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<p id=\"toc\"></p>\n\n<br><br>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #3eb489; background-color: #ffffff;\">TABLE OF CONTENTS</h1>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#imports\">0&nbsp;&nbsp;&nbsp;&nbsp;IMPORTS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#background_information\">1&nbsp;&nbsp;&nbsp;&nbsp;BACKGROUND INFORMATION</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#setup\">2&nbsp;&nbsp;&nbsp;&nbsp;SETUP</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#helper_functions\">3&nbsp;&nbsp;&nbsp;&nbsp;HELPER FUNCTIONS</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#dataset_exploration\">4&nbsp;&nbsp;&nbsp;&nbsp;DATASET EXPLORATION</a></h3>\n\n---\n\n<h3 style=\"text-indent: 10vw; font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\"><a href=\"#model_baseline\">5&nbsp;&nbsp;&nbsp;&nbsp;BASELINE</a></h3>\n\n---","metadata":{"papermill":{"duration":0.085591,"end_time":"2021-11-06T21:15:09.78303","exception":false,"start_time":"2021-11-06T21:15:09.697439","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"imports\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; background-color: #ffffff; color: #3eb489;\" id=\"imports\">0&nbsp;&nbsp;IMPORTS&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>","metadata":{"papermill":{"duration":0.050527,"end_time":"2021-11-06T21:15:09.894476","exception":false,"start_time":"2021-11-06T21:15:09.843949","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"\\n... IMPORTS STARTING ...\\n\")\n\n# print(\"\\n... PIP/APT INSTALLS AND DOWNLOADS/ZIP STARTING ...\")\n# print(\"... PIP/APT INSTALLS COMPLETE ...\\n\")\n\nprint(\"\\n\\tVERSION INFORMATION\")\n# Machine Learning and Data Science Imports\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport tensorflow_addons as tfa; print(f\"\\t\\t– TENSORFLOW ADDONS VERSION: {tfa.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None;\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\");\nfrom sklearn.preprocessing import RobustScaler, PolynomialFeatures\nfrom pandarallel import pandarallel; pandarallel.initialize();\nfrom sklearn.model_selection import GroupKFold, StratifiedKFold\n\n# Built In Imports\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom glob import glob\nimport warnings\nimport requests\nimport hashlib\nimport imageio\nimport IPython\nimport sklearn\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport os\nimport gc\nimport re\n\n# Visualization Imports\nfrom matplotlib.colors import ListedColormap\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nimport plotly.express as px\nimport seaborn as sns\nfrom PIL import Image, ImageEnhance\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom matplotlib import animation, rc; rc('animation', html='jshtml')\nimport plotly\nimport PIL\nimport cv2\n\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\n    \nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{"papermill":{"duration":162.144149,"end_time":"2021-11-06T21:17:52.087371","exception":false,"start_time":"2021-11-06T21:15:09.943222","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-02-11T15:14:38.902216Z","iopub.execute_input":"2022-02-11T15:14:38.902547Z","iopub.status.idle":"2022-02-11T15:14:48.528490Z","shell.execute_reply.started":"2022-02-11T15:14:38.902512Z","shell.execute_reply":"2022-02-11T15:14:48.527445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n<a id=\"background_information\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #3eb489; background-color: #ffffff;\" id=\"background_information\">1&nbsp;&nbsp;BACKGROUND INFORMATION&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>\n\n---\n","metadata":{"papermill":{"duration":0.05019,"end_time":"2021-11-06T21:17:52.231372","exception":false,"start_time":"2021-11-06T21:17:52.181182","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">1.1 BASIC COMPETITION INFORMATION</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">PRIMARY TASK DESCRIPTION</b>\n\nIn this competition, you’ll develop a model to **match** individual whales and dolphins by **unique—but often subtle—characteristics** of their natural markings. \n\nYou'll pay particular attention to **dorsal fins** and **lateral body views** in image sets from a **multi-species dataset** built by 28 research institutions.\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">CONTEXT</b>\n\nWe use fingerprints and facial recognition to identify people, but can we use similar approaches with animals? In fact, researchers manually track marine life by the shape and markings on their tails, dorsal fins, heads and other body parts. Identification by natural markings via photographs—known as photo-ID—is a powerful tool for marine mammal science. It allows individual animals to be tracked over time and enables assessments of population status and trends. With your help to automate whale and dolphin photo-ID, researchers can reduce image identification times by over 99%. More efficient identification could enable a scale of study previously unaffordable or impossible.\n\nCurrently, most research institutions rely on time-intensive—and sometimes inaccurate—manual matching by the human eye. Thousands of hours go into manual matching, which involves staring at photos to compare one individual to another, finding matches, and identifying new individuals. While researchers enjoy looking at a whale photo or two, manual matching limits the scope and reach.\n\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">MORE BACKGROUND INFORMATION</b>\n\n\nTBD","metadata":{"papermill":{"duration":0.053372,"end_time":"2021-11-06T21:17:52.337029","exception":false,"start_time":"2021-11-06T21:17:52.283657","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">1.2 COMPETITION EVALUATION</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">GENERAL EVALUATION INFORMATION</b>\n\n**Submissions are evaluated according to the Mean Average Precision @ 5 (MAP@5)**\n\nLATEX TO BE INSERTED\n\n*where 𝑈 is the number of images, 𝑃(𝑘) is the precision at cutoff 𝑘, 𝑛 is the number predictions per image, and 𝑟𝑒𝑙(𝑘) is an indicator function equaling 1 if the item at rank 𝑘 is a relevant (correct) label, zero otherwise.*\n\nOnce a correct label has been scored for an observation, that label is no longer considered relevant for that observation, and additional predictions of that label are skipped in the calculation. For example, if the correct label is A for an observation, the following predictions all score an average precision of 1.0.\n\n```\n[A, B, C, D, E]\n[A, A, A, A, A]\n[A, B, A, C, A]\n```\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">SUBMISSION FORMAT</b>\n\nFor each image in the test set, you may predict up to 5 **`individual_id`** labels. There are individuals in the test set that are not seen in the training data; these should be predicted as new_individual. The file should contain a header and have the following format:\n\n```\nimage,predictions \n000188a72f2562.jpg,37c7aba965a5 114207cab555 a6e325d8e924 19fbb960f07d new_individual \n000ba09273d6f3.jpg,37c7aba965a5 114207cab555 a6e325d8e924 19fbb960f07d new_individual \n...\n``` \n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase; color: red;\">IS THIS A CODE COMPETITION?</b>\n\n<font style=\"color:red; font-weight: bold; font-size: 20px;\">NO!</font>\n","metadata":{"papermill":{"duration":0.052259,"end_time":"2021-11-06T21:17:52.443218","exception":false,"start_time":"2021-11-06T21:17:52.390959","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">1.3 DATASET OVERVIEW</h3>\n\n---\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">GENERAL INFORMATION</b>\n\nIn the [**previous HappyWhale competition**](https://www.kaggle.com/c/humpback-whale-identification), the task was to predict individual humpback whales from images of their flukes. Whales and dolphins in this dataset can be identified by shapes, features and markings (some natural, some acquired) of dorsal fins, backs, heads and flanks. Some species and some individuals have highly distinct features, others are very much less distinct. Further, individual features may change over time. \n\nThis competition expands that task significantly: \n* data in this competition contains images of over **15,000 unique individual marine mammals from 30 different species collected from 28 different research organizations**. \n* Individuals have been manually identified and given an **`individual_id`** by marine researches\n\nYour task is to correctly identify these individuals in the images. It's a challenging task that has the potential to drive significant advancements in understanding and protecting marine mammals across the globe.\n\nAn important note about data quality: \n* Bringing together this dataset from many different research organization posed a number of practical challenges. \n* Significant effort has been made to minimize data quality issues and as well as to minimize leakage as much as possible.\n* There are undoubtably issues. \n* We encourage the community to report these things so that future versions of the data can be improved, but unless there is a significant issue, we don't expect to make updates to the data during the competition.\n\n<br><b style=\"text-decoration: underline; font-family: Verdana; text-transform: uppercase;\">DISCOVERED  INFORMATION [TENTATIVE]</b>\n\nTBD","metadata":{"papermill":{"duration":0.052579,"end_time":"2021-11-06T21:17:52.548406","exception":false,"start_time":"2021-11-06T21:17:52.495827","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<br>\n\n<a id=\"background_information\"></a>\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #3eb489; background-color: #ffffff;\" id=\"setup\">2&nbsp;&nbsp;SETUP&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a></h1>\n\n---\n","metadata":{"papermill":{"duration":0.052944,"end_time":"2021-11-06T21:17:52.65284","exception":false,"start_time":"2021-11-06T21:17:52.599896","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">2.1 ACCELERATOR DETECTION</h3>\n\n---\n\nIn order to use **`TPU`**, we use **`TPUClusterResolver`** for the initialization which is necessary to connect to the remote cluster and initialize cloud TPUs. Let's go over two important points\n\n1. When using TPU on Kaggle, you don't need to specify arguments for **`TPUClusterResolver`**\n2. However, on **G**oogle **C**ompute **E**ngine (**GCE**), you will need to do the following:\n\n<br>\n\n```python\n# The name you gave to the TPU to use\nTPU_WORKER = 'my-tpu-name'\n\n# or you can also specify the grpc path directly\n# TPU_WORKER = 'grpc://xxx.xxx.xxx.xxx:8470'\n\n# The zone you chose when you created the TPU to use on GCP.\nZONE = 'us-east1-b'\n\n# The name of the GCP project where you created the TPU to use on GCP.\nPROJECT = 'my-tpu-project'\n\ntpu = tf.distribute.cluster_resolver.TPUClusterResolver(tpu=TPU_WORKER, zone=ZONE, project=PROJECT)\n```\n\n<div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">🛑 &nbsp; WARNING:</b><br><br>- Although the Tensorflow documentation says it is the <b>project name</b> that should be provided for the argument <b><code>`project`</code></b>, it is actually the <b>Project ID</b>, that you should provide. This can be found on the GCP project dashboard page.<br>\n</div>\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📖 &nbsp; REFERENCES:</b><br><br>\n    - <a href=\"https://www.tensorflow.org/guide/tpu#tpu_initialization\"><b>Guide - Use TPUs</b></a><br>\n    - <a href=\"https://www.tensorflow.org/api_docs/python/tf/distribute/cluster_resolver/TPUClusterResolver\"><b>Doc - TPUClusterResolver</b></a><br>\n\n</div>","metadata":{"papermill":{"duration":0.053821,"end_time":"2021-11-06T21:17:52.761303","exception":false,"start_time":"2021-11-06T21:17:52.707482","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(f\"\\n... ACCELERATOR SETUP STARTING ...\\n\")\n\n# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    TPU = tf.distribute.cluster_resolver.TPUClusterResolver()  \nexcept ValueError:\n    TPU = None\n\nif TPU:\n    print(f\"\\n... RUNNING ON TPU - {TPU.master()}...\")\n    tf.config.experimental_connect_to_cluster(TPU)\n    tf.tpu.experimental.initialize_tpu_system(TPU)\n    strategy = tf.distribute.experimental.TPUStrategy(TPU)\nelse:\n    print(f\"\\n... RUNNING ON CPU/GPU ...\")\n    # Yield the default distribution strategy in Tensorflow\n    #   --> Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy() \n\n# What Is a Replica?\n#    --> A single Cloud TPU device consists of FOUR chips, each of which has TWO TPU cores. \n#    --> Therefore, for efficient utilization of Cloud TPU, a program should make use of each of the EIGHT (4x2) cores. \n#    --> Each replica is essentially a copy of the training graph that is run on each core and \n#        trains a mini-batch containing 1/8th of the overall batch size\nN_REPLICAS = strategy.num_replicas_in_sync\n    \nprint(f\"... # OF REPLICAS: {N_REPLICAS} ...\\n\")\n\nprint(f\"\\n... ACCELERATOR SETUP COMPLTED ...\\n\")","metadata":{"papermill":{"duration":0.07574,"end_time":"2021-11-06T21:17:52.892074","exception":false,"start_time":"2021-11-06T21:17:52.816334","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-02-11T15:14:48.530240Z","iopub.execute_input":"2022-02-11T15:14:48.530573Z","iopub.status.idle":"2022-02-11T15:14:48.547330Z","shell.execute_reply.started":"2022-02-11T15:14:48.530541Z","shell.execute_reply":"2022-02-11T15:14:48.546442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">2.2 COMPETITION DATA ACCESS</h3>\n\n---\n\nTPUs read data must be read directly from **G**oogle **C**loud **S**torage **(GCS)**. Kaggle provides a utility library – **`KaggleDatasets`** – which has a utility function **`.get_gcs_path`** that will allow us to access the location of our input datasets within **GCS**.<br><br>\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📌 &nbsp; TIPS:</b><br><br>- If you have multiple datasets attached to the notebook, you should pass the name of a specific dataset to the <b><code>`get_gcs_path()`</code></b> function. <i>In our case, the name of the dataset is the name of the directory the dataset is mounted within.</i><br><br>\n</div>","metadata":{"papermill":{"duration":0.053551,"end_time":"2021-11-06T21:17:52.999073","exception":false,"start_time":"2021-11-06T21:17:52.945522","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"\\n... DATA ACCESS SETUP STARTED ...\\n\")\n\nif TPU:\n    # Google Cloud Dataset path to training and validation images\n    DATA_DIR = KaggleDatasets().get_gcs_path('happy-whale-and-dolphin')\n    save_locally = tf.saved_model.SaveOptions(experimental_io_device='/job:localhost')\nelse:\n    # Local path to training and validation images\n    DATA_DIR = \"/kaggle/input/happy-whale-and-dolphin\"\n    save_locally = None\n\nEXTRA_DATA_DIR = \"/kaggle/input/extra-happywhale-metadata\"\n    \nprint(f\"\\n... DATA DIRECTORY PATH IS:\\n\\t--> {DATA_DIR}\")\nprint(f\"\\n... EXTRA METADATA DIRECTORY PATH IS:\\n\\t--> {EXTRA_DATA_DIR}\")\n\nprint(f\"\\n... IMMEDIATE CONTENTS OF DATA DIRECTORY IS:\")\nfor file in tf.io.gfile.glob(os.path.join(DATA_DIR, \"*\")): print(f\"\\t--> {file}\")\n\nprint(f\"\\n... IMMEDIATE CONTENTS OF EXTRA METADATA DIRECTORY IS:\")\nfor file in tf.io.gfile.glob(os.path.join(EXTRA_DATA_DIR, \"*\")): print(f\"\\t--> {file}\")\n\nprint(\"\\n\\n... DATA ACCESS SETUP COMPLETED ...\\n\")","metadata":{"papermill":{"duration":0.0797,"end_time":"2021-11-06T21:17:53.133972","exception":false,"start_time":"2021-11-06T21:17:53.054272","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-02-11T15:14:48.548971Z","iopub.execute_input":"2022-02-11T15:14:48.549291Z","iopub.status.idle":"2022-02-11T15:14:48.581306Z","shell.execute_reply.started":"2022-02-11T15:14:48.549247Z","shell.execute_reply":"2022-02-11T15:14:48.580624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">2.3 LEVERAGING XLA OPTIMIZATIONS</h3>\n\n---\n\n\n**XLA** (Accelerated Linear Algebra) is a domain-specific compiler for linear algebra that can accelerate TensorFlow models with potentially no source code changes. **The results are improvements in speed and memory usage**.\n\n<br>\n\nWhen a TensorFlow program is run, all of the operations are executed individually by the TensorFlow executor. Each TensorFlow operation has a precompiled GPU/TPU kernel implementation that the executor dispatches to.\n\nXLA provides us with an alternative mode of running models: it compiles the TensorFlow graph into a sequence of computation kernels generated specifically for the given model. Because these kernels are unique to the model, they can exploit model-specific information for optimization.<br><br>\n\n<div class=\"alert alert-block alert-danger\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">🛑 &nbsp; WARNING:</b><br><br>- XLA can not currently compile functions where dimensions are not inferrable: that is, if it's not possible to infer the dimensions of all tensors without running the entire computation<br>\n</div>\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📌 &nbsp; NOTE:</b><br><br>- XLA compilation is only applied to code that is compiled into a graph (in <b>TF2</b> that's only a code inside <b><code>tf.function</code></b>).<br>- The <b><code>jit_compile</code></b> API has must-compile semantics, i.e. either the entire function is compiled with XLA, or an <b><code>errors.InvalidArgumentError</code></b> exception is thrown)\n</div>\n\n<div class=\"alert alert-block alert-info\" style=\"margin: 2em; line-height: 1.7em; font-family: Verdana;\">\n    <b style=\"font-size: 16px;\">📖 &nbsp; REFERENCE:</b><br><br>    - <a href=\"https://www.tensorflow.org/xla\"><b>XLA: Optimizing Compiler for Machine Learning</b></a><br>\n</div>","metadata":{"papermill":{"duration":0.051494,"end_time":"2021-11-06T21:17:53.236017","exception":false,"start_time":"2021-11-06T21:17:53.184523","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(f\"\\n... XLA OPTIMIZATIONS STARTING ...\\n\")\n\nprint(f\"\\n... CONFIGURE JIT (JUST IN TIME) COMPILATION ...\\n\")\n# enable XLA optmizations (10% speedup when using @tf.function calls)\ntf.config.optimizer.set_jit(True)\n\nprint(f\"\\n... XLA OPTIMIZATIONS COMPLETED ...\\n\")","metadata":{"papermill":{"duration":0.128128,"end_time":"2021-11-06T21:17:53.442803","exception":false,"start_time":"2021-11-06T21:17:53.314675","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-02-11T15:14:48.582592Z","iopub.execute_input":"2022-02-11T15:14:48.582859Z","iopub.status.idle":"2022-02-11T15:14:48.591486Z","shell.execute_reply.started":"2022-02-11T15:14:48.582827Z","shell.execute_reply":"2022-02-11T15:14:48.590804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h3 style=\"font-family: Verdana; font-size: 20px; font-style: normal; font-weight: normal; text-decoration: none; text-transform: none; letter-spacing: 2px; color: #3eb489; background-color: #ffffff;\">2.4 BASIC DATA DEFINITIONS & INITIALIZATIONS</h3>\n\n---\n","metadata":{"papermill":{"duration":0.090507,"end_time":"2021-11-06T21:17:53.624612","exception":false,"start_time":"2021-11-06T21:17:53.534105","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"\\n... BASIC DATA SETUP STARTING ...\\n\\n\")\n\n# Thanks @karthickp6 & @kwentar and others for noticing this\nFIX_NAME_MAPPING = {\"bottlenose_dolpin\":\"bottlenose_dolphin\", \n                    \"kiler_whale\":\"killer_whale\",\n                    \"pilot_whale\":\"short_finned_pilot_whale\",\n                    \"globis\":\"short_finned_pilot_whale\"}\n\n\nprint(\"\\n... TRAIN DATAFRAME ...\\n\")\nTRAIN_CSV = os.path.join(DATA_DIR, \"train.csv\")\nEX_META_TRAIN_CSV = os.path.join(EXTRA_DATA_DIR, \"train.csv\")\ntrain_df = pd.read_csv(TRAIN_CSV)\n\nex_train_df = pd.read_csv(EX_META_TRAIN_CSV)\ntrain_df[\"img_path\"] = os.path.join(DATA_DIR, \"train_images\")+\"/\"+train_df.image\ntrain_df[\"img_shape\"] = ex_train_df[\"img_shape\"]\ntrain_df[\"img_width\"] = ex_train_df[\"img_width\"]\ntrain_df[\"img_height\"] = ex_train_df[\"img_height\"]\ntrain_df[\"species\"] = train_df[\"species\"].apply(lambda x: x if x not in FIX_NAME_MAPPING.keys() else FIX_NAME_MAPPING[x])\ntrain_df[\"n_img_of_ind\"] = train_df.individual_id.map(train_df.individual_id.value_counts().to_dict())\n\nall_species = sorted(train_df.species.unique().tolist())\nspecies_int2str_lbl_map = {i:_s for i,_s in enumerate(all_species)}\nspecies_str2int_lbl_map = {v:k for k,v in species_int2str_lbl_map.items()}\nN_SPECIES = len(all_species)\n\nall_individuals = sorted(train_df.individual_id.unique().tolist())\nind_int2str_lbl_map = {i:_s for i,_s in enumerate(all_individuals)}\nind_str2int_lbl_map = {v:k for k,v in ind_int2str_lbl_map.items()}\nN_IND = len(all_individuals)\n\n\ntrain_df[\"ind_sparse_lbl\"] = train_df[\"individual_id\"].map(ind_str2int_lbl_map)\ntrain_df[\"species_sparse_lbl\"] = train_df[\"species\"].map(species_str2int_lbl_map)\ntrain_df = train_df.sample(len(train_df)).reset_index() # shuffle\n\ndisplay(train_df)\n\nprint(\"\\n... TEST DATAFRAME ...\\n\")\nTEST_CSV = os.path.join(DATA_DIR, \"sample_submission.csv\")\nEX_META_TEST_CSV = os.path.join(EXTRA_DATA_DIR, \"test.csv\")\ntest_df = pd.read_csv(TEST_CSV)\nex_test_df = pd.read_csv(EX_META_TEST_CSV)\ntest_df[\"img_path\"] = os.path.join(DATA_DIR, \"test_images\")+\"/\"+test_df.image\ntest_df[\"img_shape\"] = ex_test_df[\"img_shape\"]\ntest_df[\"img_width\"] = ex_test_df[\"img_width\"]\ntest_df[\"img_height\"] = ex_test_df[\"img_height\"]\ntest_df = test_df.drop(columns=[\"predictions\"])\ndisplay(test_df)\n\nprint(\"\\n... SS DATAFRAME ..\\n\")\nSS_CSV = os.path.join(DATA_DIR, \"sample_submission.csv\")\nss_df = pd.read_csv(SS_CSV)\ndisplay(ss_df)\n\nprint(\"\\n\\n... BASIC DATA SETUP FINISHING ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2022-02-11T15:15:40.510964Z","iopub.execute_input":"2022-02-11T15:15:40.511271Z","iopub.status.idle":"2022-02-11T15:15:41.188919Z","shell.execute_reply.started":"2022-02-11T15:15:40.511237Z","shell.execute_reply":"2022-02-11T15:15:41.187883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n\n<a id=\"helper_functions\"></a>\n\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #3eb489; background-color: #ffffff;\" id=\"helper_functions\">\n    3&nbsp;&nbsp;HELPER FUNCTION & CLASSES&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a>\n</h1>\n\n---","metadata":{"papermill":{"duration":0.054893,"end_time":"2021-11-06T21:17:54.695576","exception":false,"start_time":"2021-11-06T21:17:54.640683","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def flatten_l_o_l(nested_list):\n    \"\"\" Flatten a list of lists \"\"\"\n    return [item for sublist in nested_list for item in sublist]\n\ndef tf_load_image(img_path, reshape_to=(512,880), cast_to_unit8=True):\n    _img = tf.image.resize(tf.image.decode_image(tf.io.read_file(img_path), channels=3, expand_animations=False), reshape_to, method=\"bicubic\")\n    if cast_to_unit8:\n        return tf.cast(_img, tf.uint8)\n    else:\n        return _img\n\ndef _bytes_feature(value, is_list=False):\n    \"\"\"Returns a bytes_list from a string / byte.\"\"\"\n    if isinstance(value, type(tf.constant(0))):\n        value = value.numpy() # BytesList won't unpack a string from an EagerTensor.\n    \n    if not is_list:\n        value = [value]\n    \n    return tf.train.Feature(bytes_list=tf.train.BytesList(value=value))\n\ndef _float_feature(value, is_list=False):\n    \"\"\"Returns a float_list from a float / double.\"\"\"\n        \n    if not is_list:\n        value = [value]\n        \n    return tf.train.Feature(float_list=tf.train.FloatList(value=value))\n\ndef _int64_feature(value, is_list=False):\n    \"\"\"Returns an int64_list from a bool / enum / int / uint.\"\"\"\n        \n    if not is_list:\n        value = [value]\n        \n    return tf.train.Feature(int64_list=tf.train.Int64List(value=value))\n\ndef create_tf_dataset(paths, gen_ids, labels=None, ind_ids=None):\n    ds = tf.data.Dataset.from_tensor_slices(paths)\n    ds_ids = tf.data.Dataset.from_tensor_slices(gen_ids)\n    \n    if labels is None:\n        ds = tf.data.Dataset.zip((ds, ds_ids))\n        ds = ds.map(lambda x,y: (tf_load_image(x), y))\n        return ds\n    else:\n        ds_ind_ids = tf.data.Dataset.from_tensor_slices(ind_ids)\n        ds_label = tf.data.Dataset.from_tensor_slices(tf.constant(labels, dtype=tf.uint8))\n        ds = tf.data.Dataset.zip((ds, ds_ids, ds_label, ds_ind_ids))\n        ds = ds.map(lambda x,y,z,a: (tf_load_image(x), y, z, a))        \n        return ds\n    \n    \ndef serialize_raw(image, image_id, label=None, ind_id=None):\n    \"\"\"\n    Creates a tf.Example message ready to be written to a file from 4 features.\n\n    Args:\n        image (TBD): TBD\n        image_id (str): TBD\n        target (str): | delimited integers\n    \n    Returns:\n        A tf.Example Message ready to be written to file\n    \"\"\"\n    \n    # Create a dictionary mapping the feature name to the \n    # tf.Example-compatible data type.\n    feature = {'image': _bytes_feature(tf.io.encode_jpeg(image), is_list=False)}\n    \n    if label is not None:\n        feature[\"label\"] = _int64_feature(label, is_list=False)\n        feature[\"ind_id\"] = _int64_feature(ind_id, is_list=False)\n    \n    feature[\"image_id\"] = _bytes_feature(image_id, is_list=False)\n        \n    # Create a Features message using tf.train.Example.\n    example_proto = tf.train.Example(features=tf.train.Features(feature=feature))\n    return example_proto.SerializeToString()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T15:15:55.191194Z","iopub.execute_input":"2022-02-11T15:15:55.191514Z","iopub.status.idle":"2022-02-11T15:15:55.211777Z","shell.execute_reply.started":"2022-02-11T15:15:55.191480Z","shell.execute_reply":"2022-02-11T15:15:55.211061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<br>\n\n\n<a id=\"dataset_exploration\"></a>\n\n\n<h1 style=\"font-family: Verdana; font-size: 24px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; letter-spacing: 3px; color: #3eb489; background-color: #ffffff;\" id=\"dataset_exploration\">\n    4&nbsp;&nbsp;DATASET PREPERATION&nbsp;&nbsp;&nbsp;&nbsp;<a href=\"#toc\">&#10514;</a>\n</h1>\n\n---","metadata":{"papermill":{"duration":0.056443,"end_time":"2021-11-06T21:17:54.960518","exception":false,"start_time":"2021-11-06T21:17:54.904075","status":"completed"},"tags":[]}},{"cell_type":"code","source":"K_FOLDS = 10\nskfold = StratifiedKFold(n_splits=10)\nsub_train_dfs = []\nfor train_idxs, val_idxs in skfold.split(train_df[\"img_path\"], train_df[\"individual_id\"]):\n    sub_train_df = train_df.iloc[val_idxs].reset_index()\n    N_TRAIN = len(sub_train_df)\n    sub_train_dfs.append(sub_train_df)\n    \nprint(\"# OF TRAIN :\", N_TRAIN)\n\nfor sub_train_df in sub_train_dfs:\n    print(\"\\n\\n\\n------------------------\\n\")\n    for i, (k,v) in enumerate(sub_train_df.species.value_counts().items()):\n        print(f\"\\n ----- {k} - ({i+1}) -----\")\n        print(f\"TRAIN : %{100*v/N_TRAIN:.1f}\")\n\ntrain_unique_individuals = set(sub_train_df.individual_id.unique().tolist())","metadata":{"execution":{"iopub.status.busy":"2022-02-11T15:17:49.151234Z","iopub.execute_input":"2022-02-11T15:17:49.152137Z","iopub.status.idle":"2022-02-11T15:17:49.978311Z","shell.execute_reply.started":"2022-02-11T15:17:49.152083Z","shell.execute_reply":"2022-02-11T15:17:49.977214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_datasets = []\n\nfor t_df in sub_train_dfs:\n    train_datasets.append(create_tf_dataset(paths   = t_df.img_path.tolist(), \n                                            gen_ids = t_df.image.apply(lambda x: x[:-4]).tolist(), \n                                            labels  = t_df.species_sparse_lbl.tolist(), \n                                            ind_ids = t_df.ind_sparse_lbl.tolist()))\ntest_ds = create_tf_dataset(paths   = test_df.img_path.tolist(), \n                            gen_ids = test_df.image.apply(lambda x: x[:-4]).tolist())\n\n# print(train_datasets)\n# for x,y in train_datasets[0].take(1):\n#     plt.imshow(x[0]/255.)\n#     plt.show()\n\n# print(val_datasets)\n# for x,y in val_datasets[0].take(1):\n#     plt.imshow(x[0]/255.)\n#     plt.show()\n\n# print(test_ds)\n# for x,y in test_ds.take(1):\n#     plt.imshow(x/255.)\n#     plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-11T15:18:32.492560Z","iopub.execute_input":"2022-02-11T15:18:32.492965Z","iopub.status.idle":"2022-02-11T15:18:33.808889Z","shell.execute_reply.started":"2022-02-11T15:18:32.492930Z","shell.execute_reply":"2022-02-11T15:18:33.807839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def write_tfrecords(ds, n_ex, n_ex_per_rec=1000, serialize_fn=serialize_raw, out_dir=\"/kaggle/working/512x880\", ds_type=\"train\", fold=None):\n    \"\"\"\"\"\"\n    n_recs = int(np.ceil(n_ex/n_ex_per_rec))\n    \n    # Make dataset iterable\n    ds = iter(ds)\n        \n    out_dir = os.path.join(out_dir, ds_type)\n    # Create folder\n    if not os.path.isdir(out_dir):\n        os.makedirs(out_dir, exist_ok=True)\n        \n    # Create tfrecords\n    for i in tqdm(range(n_recs), total=n_recs):\n        print(f\"\\n... Writing {ds_type.title()} TFRecord {i+1} of {n_recs} For Fold #{fold}...\\n\")\n        if fold is not None:\n            tfrec_path = os.path.join(out_dir, f\"{ds_type}_fold_{fold}__{(i+1):02}_{n_recs:02}.tfrec\")\n        else:\n            tfrec_path = os.path.join(out_dir, f\"{ds_type}__{(i+1):02}_{n_recs:02}.tfrec\")\n        \n        # This makes the tfrecord\n        with tf.io.TFRecordWriter(tfrec_path) as writer:\n            for ex in tqdm(range(n_ex_per_rec), total=n_ex_per_rec):\n                try:\n                    example = serialize_fn(*next(ds))\n                    writer.write(example)\n                except:\n                    break\n\nfor n_fold, (_ds, _df) in enumerate(zip(train_datasets, sub_train_dfs)):\n    write_tfrecords(_ds, len(_df), fold=n_fold+1)\n    \nwrite_tfrecords(test_ds, len(test_df), ds_type=\"test\")","metadata":{"execution":{"iopub.status.busy":"2022-02-11T15:19:05.577272Z","iopub.execute_input":"2022-02-11T15:19:05.577678Z","iopub.status.idle":"2022-02-11T15:49:07.760445Z","shell.execute_reply.started":"2022-02-11T15:19:05.577637Z","shell.execute_reply":"2022-02-11T15:49:07.757681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, sub_train_df in enumerate(sub_train_dfs):\n    sub_train_df.to_csv(f\"./train_df_fold_{i+1}.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-11T15:49:07.767444Z","iopub.execute_input":"2022-02-11T15:49:07.767904Z","iopub.status.idle":"2022-02-11T15:49:08.410640Z","shell.execute_reply.started":"2022-02-11T15:49:07.767867Z","shell.execute_reply":"2022-02-11T15:49:08.409718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Decode TBD - Full encoding of all folds tbd","metadata":{}},{"cell_type":"code","source":"def decode_image(image_data, resize_to=(512,880,3)):\n    \"\"\" Function to decode the tf.string containing image information \n    \n    \n    Args:\n        image_data (tf.string): String containing encoded image data from tf.Example\n        resize_to (tuple, optional): Size that we will reshape the tensor to (required for TPU)\n        to_norm_val (int, optional): The value that will allow us to normalize the tensor to 0-1\n    \n    Returns:\n        Tensor containing the resized single-channel image in the appropriate dtype\n    \"\"\"\n    image = tf.image.decode_png(image_data, channels=3, dtype=tf.uint8)\n    image = tf.reshape(image, resize_to)\n    return tf.cast(image, tf.float32)\n\n\ndef decode_example(serialized_example, is_test=False, return_ind_id=False, return_img_id=False, img_shape=(512,880,3)):\n    \"\"\" Parses a set of features and label from the given `serialized_example`.\n        \n        It is used as a map function for `dataset.map`\n\n    Args:\n        serialized_example (tf.Example): A serialized example containing the\n            following features:\n                – 'image'\n                – 'image_id'\n                – 'label'\n        is_test (bool, optional): Whether to allow for the label feature\n        \n    Returns:\n        A decoded tf.data.Dataset object representing the tfrecord dataset\n    \"\"\"\n    feature_dict = {\n        'image': tf.io.FixedLenFeature(shape=[], dtype=tf.string, default_value=''),\n    }\n    \n    if not is_test:\n        feature_dict[\"label\"] = tf.io.FixedLenFeature(shape=[], dtype=tf.int64, default_value=0)\n        feature_dict[\"ind_id\"] = tf.io.FixedLenFeature(shape=[], dtype=tf.int64, default_value=0)\n    feature_dict[\"image_id\"] = tf.io.FixedLenFeature(shape=[], dtype=tf.string, default_value='')\n    \n    # Define a parser\n    features = tf.io.parse_single_example(serialized_example, features=feature_dict)\n    \n    # Decode the tf.string\n    image = decode_image(features['image'], resize_to=img_shape)\n    \n    # Figure out the correct information to return\n    if is_test:\n        image_id = features[\"image_id\"] \n        return image, image_id\n    else:\n        label = features[\"label\"]\n        if (return_ind_id or return_img_id):\n            ind_id = features[\"ind_id\"]\n            image_id = features[\"image_id\"]\n            return image, label, ind_id, image_id\n        else:\n            return image, label","metadata":{"execution":{"iopub.status.busy":"2022-02-11T15:51:52.772859Z","iopub.execute_input":"2022-02-11T15:51:52.773951Z","iopub.status.idle":"2022-02-11T15:51:52.790846Z","shell.execute_reply.started":"2022-02-11T15:51:52.773901Z","shell.execute_reply":"2022-02-11T15:51:52.789700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ALL_TFRECORDS = tf.io.gfile.glob(os.path.join(\"/kaggle/working/512x880\", \"**/*.tfrec\"))\ntfrecord_map = {\n    \"train\":{f\"fold_{i}\":[_tfrec for _tfrec in ALL_TFRECORDS if f\"train_fold_{i}\" in _tfrec] for i in range(1, K_FOLDS+1)},\n    \"test\":[_tfrec for _tfrec in ALL_TFRECORDS if \"test__\" in _tfrec]\n}\n\ntfds_map = {\"train\":{}, \"test\":[]}\nfor k,v in tfrecord_map.items():\n    if k in [\"train\", \"val\"]:\n        for i in range(1,K_FOLDS+1):\n            tfds_map[k][f\"fold_{i}\"]=tf.data.TFRecordDataset(v[f\"fold_{i}\"], num_parallel_reads=tf.data.AUTOTUNE)\n    else:\n        tfds_map[k] = tf.data.TFRecordDataset(v, num_parallel_reads=tf.data.AUTOTUNE)\n\nprint(\"\\n... DECODING RAW TFRECORD DATASETS STARTING ...\\n\")\n\nfor k,v in tfds_map.items():\n    if k in [\"train\", \"val\"]:\n        for i in range(1,K_FOLDS+1):\n            tfds_map[k][f\"fold_{i}\"]=v[f\"fold_{i}\"].map(lambda x: decode_example(x, is_test=False, return_img_id=True), num_parallel_calls=tf.data.AUTOTUNE)\n    else:\n        tfds_map[k]=v.map(lambda x: decode_example(x, is_test=True), num_parallel_calls=tf.data.AUTOTUNE)\n\nprint(f\"\\n... THE DECODED TF.DATA.TFRECORDDATASET OBJECT:\" \\\n      f\"\\n\\t--> ((image), (image_id - optional), (ind_id - optional), (label))\" \\\n      f\"\\n\\t--> {tfds_map['train'][f'fold_1']}\\n\")\n\nprint(\"\\n... 4 EXAMPLES OF TRAIN IMAGES AND LABELS AFTER DECODING AND AUGMENTATION ...\")\nfor i, (img, label, ind, img_id) in enumerate(tfds_map[\"train\"][f\"fold_1\"].take(2)):\n    display(train_df[train_df.image==f\"{img_id.numpy().decode()}.jpg\"])\n    print(f\"\\nIMAGE SHAPE : {img.shape}\")\n    print(f\"SPECIES LABEL : {species_int2str_lbl_map[label.numpy()]}\")\n    print(f\"INDIVIDUAL LABEL: {ind_int2str_lbl_map[ind.numpy()]}\")\n    print(f\"IMAGE ID: {img_id.numpy().decode()}\")\n    plt.figure(figsize=(10,10))\n    plt.imshow(img.numpy()/255., cmap=\"gray\")\n    plt.axis(False)\n    plt.tight_layout()\n    plt.show()\n    \nprint(\"\\n... DECODING RAW TFRECORD DATASETS COMPLETED ...\\n\")\n","metadata":{"execution":{"iopub.status.busy":"2022-02-11T15:53:30.589408Z","iopub.execute_input":"2022-02-11T15:53:30.590591Z","iopub.status.idle":"2022-02-11T15:53:32.219021Z","shell.execute_reply.started":"2022-02-11T15:53:30.590527Z","shell.execute_reply":"2022-02-11T15:53:32.217960Z"},"trusted":true},"execution_count":null,"outputs":[]}]}