{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.7.6"},"papermill":{"duration":1456.900285,"end_time":"2020-11-05T09:08:00.38091","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2020-11-05T08:43:43.480625","version":"2.1.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"34fc937d2fb74b5b9b04ce50715d9701":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_e1243ded819140798c9c4350014395eb","IPY_MODEL_46a589075fac4ce2b3a704a5123bc04f"],"layout":"IPY_MODEL_cc822b1733df4e849c081b11422f2347"}},"46a589075fac4ce2b3a704a5123bc04f":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_f4f05a5e838f47e0aba20689fece89bf","placeholder":"​","style":"IPY_MODEL_bbbe439f21a6419f84d0c0b1c116f6ae","value":" 95.8M/95.8M [00:02&lt;00:00, 44.9MB/s]"}},"995c98179fef4f018a4cbd43c4678857":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":"initial"}},"a3403c59407d4df0a72d5fa5f8765b02":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"bbbe439f21a6419f84d0c0b1c116f6ae":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"cc822b1733df4e849c081b11422f2347":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e1243ded819140798c9c4350014395eb":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"100%","description_tooltip":null,"layout":"IPY_MODEL_a3403c59407d4df0a72d5fa5f8765b02","max":100441675,"min":0,"orientation":"horizontal","style":"IPY_MODEL_995c98179fef4f018a4cbd43c4678857","value":100441675}},"f4f05a5e838f47e0aba20689fece89bf":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}}},"version_major":2,"version_minor":0}},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19990,"databundleVersionId":1472735,"sourceType":"competition"}],"dockerImageVersionId":30775,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Lyft: Understanding the data  and EDA","metadata":{"papermill":{"duration":0.080767,"end_time":"2020-11-05T08:43:48.021827","exception":false,"start_time":"2020-11-05T08:43:47.94106","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"![](https://neurohive.io/wp-content/uploads/2019/07/Screenshot-from-2019-07-25-01-30-57.png)","metadata":{"papermill":{"duration":0.076493,"end_time":"2020-11-05T08:43:48.175506","exception":false,"start_time":"2020-11-05T08:43:48.099013","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### Credits:\n\n**https://www.kaggle.com/t3nyks/lyft-working-with-map-api**<br>\n**https://www.kaggle.com/jpbremer/lyft-scene-visualisations**<br>\n**https://www.kaggle.com/pestipeti/pytorch-baseline-train**","metadata":{"papermill":{"duration":0.07674,"end_time":"2020-11-05T08:43:48.331153","exception":false,"start_time":"2020-11-05T08:43:48.254413","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"This new Lyft competition is tasking us, the participants, to predict the motion of external cars, cyclists, pedestrians etc. to assist self-driving cars. This is a step ahead from last year's competition, where we were tasked with detecting three-dimensional objects, like stop signs, to teach AVs how to recognize these. ","metadata":{"papermill":{"duration":0.081397,"end_time":"2020-11-05T08:43:48.494642","exception":false,"start_time":"2020-11-05T08:43:48.413245","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"**TIP: Use plt.imshow instead of IPython.display**","metadata":{"papermill":{"duration":0.055078,"end_time":"2020-11-05T08:43:48.629646","exception":false,"start_time":"2020-11-05T08:43:48.574568","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"This is apparently the **largest collection of traffic agent motion data.** The files are stored in the .zarr file format with Python, which we can easily load using the Level 5 Kit (l5kit for the pip package). Within our training ZARRs, we have the agents, the masks for agents, frames and scenes (which you might recollect from last year) and traffic light faces.\n\nThe test ZARR however is almost practically the same format, but the only exclusion is that of the data masks. for the agents. ","metadata":{"papermill":{"duration":0.054374,"end_time":"2020-11-05T08:43:48.741042","exception":false,"start_time":"2020-11-05T08:43:48.686668","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Get started with the data","metadata":{"papermill":{"duration":0.055018,"end_time":"2020-11-05T08:43:48.850491","exception":false,"start_time":"2020-11-05T08:43:48.795473","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Wait! Before we directly get into the fancy visualization and all it entails, why not watch a short YouTube video to enlighten us on the subject of operating an autonomous vehicle?","metadata":{"papermill":{"duration":0.054405,"end_time":"2020-11-05T08:43:48.959395","exception":false,"start_time":"2020-11-05T08:43:48.90499","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from IPython.display import HTML\nHTML('<center><iframe width=\"700\" height=\"400\" src=\"https://www.youtube.com/embed/tlThdr3O5Qo?rel=0&amp;controls=0&amp;showinfo=0\" frameborder=\"0\" allowfullscreen></iframe></center>')","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:43:49.076293Z","iopub.status.busy":"2020-11-05T08:43:49.075366Z","iopub.status.idle":"2020-11-05T08:43:49.079428Z","shell.execute_reply":"2020-11-05T08:43:49.079952Z"},"papermill":{"duration":0.06599,"end_time":"2020-11-05T08:43:49.080133","exception":false,"start_time":"2020-11-05T08:43:49.014143","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Seems like this car casually handles all the normal challenges a driver faces, and that too with remarkable accuracy. Over here, we are tasked with somethign to faciliate this sort of thing - **predicting the motion of extraneous vehicles and based on that, predicting the motion path of an AV.**  To predict the motion of these extraneous factors, there are many approaches which I shall discuss later, but for now let's dive in.\n\nHere's a brief FAQ section about the dataset and all it entails:\n\n**What is the structure of the dataset?**<br>\nThe dataset is structured as follows:\n```\naerial_map\nscenes\nsemantic_map\n```\n\nwhere each scene contains roughly a minute or so of information about the motion of several extraneous vehicles and the corresponding movement of the AV.\n\nUnder scenes, we have:\n```\nsample.zarr\ntest.zarr\ntrain.zarr\nvalidate.zarr\n```\n\nNow this ZARR format is a little bit interesting, as I am willing to fathom a guess most of the participants have never worked with these. Fear not, for they are very much interoperable with NumPy and the Lyft Level 5 Kit also gives us easy ways to handle the processing of the data. Of course, there also might be a few ways to use a Pandas DataFrame in the process, which leads up a road for LightGBM.\n\nThe train.zarr contains the agents, the mask for the agents, the frames, the scenes and the traffic light faces, which I'll go into more depth later.","metadata":{"papermill":{"duration":0.05574,"end_time":"2020-11-05T08:43:49.206054","exception":false,"start_time":"2020-11-05T08:43:49.150314","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"<img src=\"https://self-driving.lyft.com/wp-content/uploads/2020/06/dataset-steps-longimg.png\"></img>","metadata":{"papermill":{"duration":0.055078,"end_time":"2020-11-05T08:43:49.31642","exception":false,"start_time":"2020-11-05T08:43:49.261342","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Now, as we can see here, we have a sensor input (in the form of last year's sensory data) and they had to detect traffic agents in last year's competition. Here, we follow the next two steps of the process: predicting the agent motion and mapping out a path for the autonomous vehicle. The sensor is a LIDAR sensor, which basically gives us a rough perspective of the motion on the road. The sensor then feeds the data back to Lyft, who then collects the data from multiple sensors/cars all over Palo Alto, collates the data and gives it to us.","metadata":{"papermill":{"duration":0.058189,"end_time":"2020-11-05T08:43:49.429551","exception":false,"start_time":"2020-11-05T08:43:49.371362","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"We may now import the lyft level 5 kit, and all that comes with it. We have quite a lot of imports and installations required...","metadata":{"papermill":{"duration":0.055036,"end_time":"2020-11-05T08:43:49.542302","exception":false,"start_time":"2020-11-05T08:43:49.487266","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"**UPDATE: Finally got GPU to work by manually installing everything, adding utility scripts did not help at all. Took a painfully long time to get myself to realize that everything needs to be done in the kernel or things will break.**","metadata":{"papermill":{"duration":0.055864,"end_time":"2020-11-05T08:43:49.65348","exception":false,"start_time":"2020-11-05T08:43:49.597616","status":"completed"},"tags":[]}},{"cell_type":"code","source":"!pip install neptune-client segmentation_models_pytorch hydra-core kekas -U -q \n!pip install --target=/kaggle/working pymap3d==2.1.0 -q\n!pip install --target=/kaggle/working strictyaml -q\n!pip install --target=/kaggle/working protobuf==3.12.2 -q\n!pip install --target=/kaggle/working transforms3d -q\n!pip install --target=/kaggle/working zarr -q\n!pip install --target=/kaggle/working ptable -q\n!pip install --no-dependencies --target=/kaggle/working l5kit==1.1.0 --upgrade -q\n!cp ../input/lyft-config-files/agent_motion_config.yaml config.yaml\nimport l5kit, os, albumentations as A\nfrom l5kit.rasterization import build_rasterizer\nfrom l5kit.configs import load_config_data\nfrom l5kit.visualization import draw_trajectory, TARGET_POINTS_COLOR\nfrom tqdm import tqdm\nfrom l5kit.geometry import transform_points\nfrom collections import Counter\nfrom l5kit.data import PERCEPTION_LABELS\nfrom prettytable import PrettyTable\nimport matplotlib.pyplot as plt\n# set env variable for data\nos.environ[\"L5KIT_DATA_FOLDER\"] = \"../input/lyft-motion-prediction-autonomous-vehicles\"\n# get config\nMONITORING = True # set this to false if you want to fork and train\nif MONITORING:\n    import utilsforlyft as U\ncfg = load_config_data(\"../input/lyft-config-files/visualisation_config.yaml\")\nimport plotly.offline as py\nimport omegaconf\npy.init_notebook_mode(connected=True)\nimport plotly.graph_objs as go\nimport plotly.tools as tls\nimport torch\nimport torch.nn.functional as F\nimport torch.nn as nn\nfrom catalyst import dl, data\nfrom catalyst.utils import metrics\nfrom torch.utils.data import DataLoader\nfrom catalyst.dl import utils, BatchOverfitCallback\nfrom torch.optim.lr_scheduler import OneCycleLR\nimport segmentation_models_pytorch as smp\nfrom catalyst.core.callbacks.early_stop import EarlyStoppingCallback\nif MONITORING:\n    from catalyst.contrib.dl.callbacks.neptune_logger import NeptuneLogger\n    from catalyst.contrib.dl.callbacks import WandbLogger\n    neptune_logger = NeptuneLogger(\n                    api_token=U.TOKEN + '=',  \n                    project_name=\"trigram19/\"+U.NAME_PROJ,\n                    offline_mode=False, \n                    name=U.NAME,\n                    params={'epoch_nr': 5}, \n                    properties={'data_source': 'lyft'},  \n                    tags=['resnet']\n                    )\n\nfrom IPython.display import display, clear_output, HTML\nimport PIL\nimport matplotlib.pyplot as plt\nfrom matplotlib import animation as ani, rc\nimport numpy as np\nimport warnings; warnings.filterwarnings('ignore')\nfrom l5kit.rasterization.rasterizer_builder import _load_metadata\nfrom kekas import Keker, DataOwner, DataKek\nfrom kekas.utils import DotDict\nfrom kekas.transformations import Transformer, to_torch, normalize","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2020-11-05T08:43:49.809285Z","iopub.status.busy":"2020-11-05T08:43:49.799049Z","iopub.status.idle":"2020-11-05T08:46:07.521025Z","shell.execute_reply":"2020-11-05T08:46:07.520483Z"},"papermill":{"duration":137.811397,"end_time":"2020-11-05T08:46:07.521157","exception":false,"start_time":"2020-11-05T08:43:49.70976","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Here we go using some helpful functions to visualize the data.","metadata":{"papermill":{"duration":0.061322,"end_time":"2020-11-05T08:46:07.646528","exception":false,"start_time":"2020-11-05T08:46:07.585206","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def plot_image(map_type, ax, agent=False):\n    cfg[\"raster_params\"][\"map_type\"] = map_type\n    rast = build_rasterizer(cfg, dm)\n    if agent:\n        dataset = AgentDataset(cfg, zarr_dataset, rast)\n    else:\n        dataset = EgoDataset(cfg, zarr_dataset, rast)\n    scene_idx = 2\n    indexes = dataset.get_scene_indices(scene_idx)\n    images = []\n    for idx in indexes:    \n        data = dataset[idx]\n        im = data[\"image\"].transpose(1, 2, 0)\n        im = dataset.rasterizer.to_rgb(im)\n        target_positions_pixels = transform_points(data[\"target_positions\"] + data[\"centroid\"][:2], data[\"world_to_image\"])\n        center_in_pixels = np.asarray(cfg[\"raster_params\"][\"ego_center\"]) * cfg[\"raster_params\"][\"raster_size\"]\n        draw_trajectory(im, target_positions_pixels, TARGET_POINTS_COLOR, yaws=data[\"target_yaws\"])\n        clear_output(wait=True)\n        ax.imshow(im[::-1])\n                \ndef animate_solution(images):\n    def animate(i):\n        im.set_data(images[i]) \n    fig, ax = plt.subplots()\n    im = ax.imshow(images[0])    \n    return ani.FuncAnimation(fig, animate, frames=len(images), interval=60)\n\ndef animation(type_):\n    cfg[\"raster_params\"][\"map_type\"] = type_\n    rast = build_rasterizer(cfg, dm)\n    dataset = EgoDataset(cfg, zarr_dataset, rast)\n    scene_idx = 34\n    indexes = dataset.get_scene_indices(scene_idx)\n    images = []\n    for idx in indexes:    \n        data = dataset[idx]\n        im = data[\"image\"].transpose(1, 2, 0)\n        im = dataset.rasterizer.to_rgb(im)\n        target_positions_pixels = transform_points(data[\"target_positions\"] + data[\"centroid\"][:2], data[\"world_to_image\"])\n        center_in_pixels = np.asarray(cfg[\"raster_params\"][\"ego_center\"]) * cfg[\"raster_params\"][\"raster_size\"]\n        draw_trajectory(im, target_positions_pixels, TARGET_POINTS_COLOR, yaws=data[\"target_yaws\"])\n        clear_output(wait=True)\n        images.append(PIL.Image.fromarray(im[::-1]))\n    anim = animate_solution(images)\n    return HTML(anim.to_jshtml())\n\ndef plot_with_tfms(tf):\n    tfms = A.Compose(tf, keypoint_params=A.KeypointParams(format=\"xy\", remove_invisible=False))\n    im = tfms(image=train_dataset_a[0]['image'], keypoints=train_dataset_a[0]['target_positions'])\n    keypoints = im['keypoints']\n    im = train_dataset_a.rasterizer.to_rgb(im['image'].transpose(1, 2, 0))\n    plt.imshow(im, cmap='Reds');\n    \n","metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-05T08:46:07.794969Z","iopub.status.busy":"2020-11-05T08:46:07.782604Z","iopub.status.idle":"2020-11-05T08:46:07.797798Z","shell.execute_reply":"2020-11-05T08:46:07.797094Z"},"papermill":{"duration":0.090453,"end_time":"2020-11-05T08:46:07.797899","exception":false,"start_time":"2020-11-05T08:46:07.707446","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now, let's get a sense of the configuration data. This will include metadata pertaining to the agents, the total time, the frames-per-scene, the scene time and the frame frequency.","metadata":{"papermill":{"duration":0.062998,"end_time":"2020-11-05T08:46:07.922907","exception":false,"start_time":"2020-11-05T08:46:07.859909","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from l5kit.data import ChunkedDataset, LocalDataManager\nfrom l5kit.dataset import EgoDataset, AgentDataset\ndm = LocalDataManager()\ndataset_path = dm.require(cfg[\"val_data_loader\"][\"key\"]);rasterizer = build_rasterizer(cfg, dm)\nzarr_dataset = ChunkedDataset(dataset_path)\ntrain_dataset_a = AgentDataset(cfg, zarr_dataset, rasterizer)\nzarr_dataset.open()\nprint(zarr_dataset)","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:46:08.057786Z","iopub.status.busy":"2020-11-05T08:46:08.056843Z","iopub.status.idle":"2020-11-05T08:46:14.162272Z","shell.execute_reply":"2020-11-05T08:46:14.163039Z"},"papermill":{"duration":6.176915,"end_time":"2020-11-05T08:46:14.163252","exception":false,"start_time":"2020-11-05T08:46:07.986337","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now, however it's time for us to look at the scenes and analyze them in depth. Theoretically, we could create a nifty little data-loader to do some heavy lifting for us.","metadata":{"papermill":{"duration":0.066401,"end_time":"2020-11-05T08:46:14.294822","exception":false,"start_time":"2020-11-05T08:46:14.228421","status":"completed"},"tags":[]}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 1, figsize=(15, 10))\nfor i, key in enumerate([\"py_semantic\"]):\n    plot_image(key, ax=axes)","metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-05T08:46:14.436812Z","iopub.status.busy":"2020-11-05T08:46:14.434925Z","iopub.status.idle":"2020-11-05T08:46:43.495719Z","shell.execute_reply":"2020-11-05T08:46:43.496633Z"},"papermill":{"duration":29.138535,"end_time":"2020-11-05T08:46:43.49684","exception":false,"start_time":"2020-11-05T08:46:14.358305","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"So, there's a lot of information in this one image. I'll try my best to point everything out, but do notify me if I make any errors. OK, let's get started with dissecting the image:\n+ We have an intersection of four roads over here.\n+ The green blob represents the AV's motion, and we would require to predict the movement of the AV in these traffic conditions as a sample.","metadata":{"papermill":{"duration":0.109696,"end_time":"2020-11-05T08:46:43.717157","exception":false,"start_time":"2020-11-05T08:46:43.607461","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"I don't exactly know what other inferences we can make without more detail on this data, so let's try a satellite-format viewing of these images. ","metadata":{"papermill":{"duration":0.066181,"end_time":"2020-11-05T08:46:43.868211","exception":false,"start_time":"2020-11-05T08:46:43.80203","status":"completed"},"tags":[]}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 1, figsize=(15, 10))\nfor i, key in enumerate([\"py_satellite\"]):\n    plot_image(key, ax=axes)","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:46:44.016628Z","iopub.status.busy":"2020-11-05T08:46:44.015621Z","iopub.status.idle":"2020-11-05T08:47:08.005691Z","shell.execute_reply":"2020-11-05T08:47:08.006283Z"},"papermill":{"duration":24.070468,"end_time":"2020-11-05T08:47:08.006442","exception":false,"start_time":"2020-11-05T08:46:43.935974","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Yes! This allows for far more detail than a simple plot without detail. I'd haphazard an educated guess, and make the following inferences:\n+ Green still represents the autonomous vehicle (AV), and blue is primarily all the other cars/vehicles/exogenous factors we need to predict for.\n+ My hypothesis is that the blue represents the path the vehicle needs to go through.\n+ If we are able to accurately predict the path the vehicles go through, it will make it easier for an AV to compute its trajectory on the fly.","metadata":{"papermill":{"duration":0.080647,"end_time":"2020-11-05T08:47:08.171328","exception":false,"start_time":"2020-11-05T08:47:08.090681","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"We also want to see how the whole charade of vehicles ","metadata":{"papermill":{"duration":0.077838,"end_time":"2020-11-05T08:47:08.329418","exception":false,"start_time":"2020-11-05T08:47:08.25158","status":"completed"},"tags":[]}},{"cell_type":"code","source":"animation(\"py_satellite\")","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:47:08.489096Z","iopub.status.busy":"2020-11-05T08:47:08.488226Z","iopub.status.idle":"2020-11-05T08:47:35.53147Z","shell.execute_reply":"2020-11-05T08:47:35.53074Z"},"papermill":{"duration":27.124916,"end_time":"2020-11-05T08:47:35.531605","exception":false,"start_time":"2020-11-05T08:47:08.406689","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"So this is a demonstration of the movement of the other vehicles and (in relation to the movement and placement of the other vehicles) the movement of the AV. The AV is currently taking only a straight path in its motion, and a straight path seems logical with the movement and placement of other vehicles.","metadata":{"papermill":{"duration":0.66923,"end_time":"2020-11-05T08:47:36.859702","exception":false,"start_time":"2020-11-05T08:47:36.190472","status":"completed"},"tags":[]}},{"cell_type":"code","source":"animation(\"py_semantic\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-05T08:47:38.533903Z","iopub.status.busy":"2020-11-05T08:47:38.532826Z","iopub.status.idle":"2020-11-05T08:48:15.126907Z","shell.execute_reply":"2020-11-05T08:48:15.126294Z"},"papermill":{"duration":37.27357,"end_time":"2020-11-05T08:48:15.127019","exception":false,"start_time":"2020-11-05T08:47:37.853449","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We're also able to take a more low-level move by using the semantic option in the Lyft level 5 kit. Or... we can go even deeper and use the box rasterizer.","metadata":{"papermill":{"duration":1.284805,"end_time":"2020-11-05T08:48:17.562963","exception":false,"start_time":"2020-11-05T08:48:16.278158","status":"completed"},"tags":[]}},{"cell_type":"code","source":"animation(\"box_debug\")","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:48:19.693561Z","iopub.status.busy":"2020-11-05T08:48:19.692565Z","iopub.status.idle":"2020-11-05T08:48:43.211899Z","shell.execute_reply":"2020-11-05T08:48:43.212518Z"},"papermill":{"duration":24.587651,"end_time":"2020-11-05T08:48:43.212677","exception":false,"start_time":"2020-11-05T08:48:18.625026","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The semantic view is good for a less clustered view but if we want a more detailed, more high-level overview of the data we should perhaps try to use the satellite view voer semantic.","metadata":{"papermill":{"duration":1.221519,"end_time":"2020-11-05T08:48:45.62179","exception":false,"start_time":"2020-11-05T08:48:44.400271","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Now, how about from the agent perspective? This would be quite interesting to consider, as we're modeling from principally the agent perspective in most public notebooks so far.","metadata":{"papermill":{"duration":1.187289,"end_time":"2020-11-05T08:48:47.999952","exception":false,"start_time":"2020-11-05T08:48:46.812663","status":"completed"},"tags":[]}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 1, figsize=(15, 10))\nfor i, key in enumerate([\"py_satellite\"]):\n    plot_image(key, ax=axes, agent=True)","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:48:50.746304Z","iopub.status.busy":"2020-11-05T08:48:50.745097Z","iopub.status.idle":"2020-11-05T08:49:36.078698Z","shell.execute_reply":"2020-11-05T08:49:36.079664Z"},"papermill":{"duration":46.793608,"end_time":"2020-11-05T08:49:36.079881","exception":false,"start_time":"2020-11-05T08:48:49.286273","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"So yes, I probably should save these as a GIF to visualize the agent movements. Let's try a simpler form of this and use the semantic view for the agent dataset.","metadata":{"papermill":{"duration":1.442934,"end_time":"2020-11-05T08:49:38.926159","exception":false,"start_time":"2020-11-05T08:49:37.483225","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Uh it seems the rasterizer renders rather well the satellite and semantic views, and both in conjunction help one to get a good sense of the positioning of each vehicle in relation to the road. You can easily understand the placement and motion of the vehicles and highway layout in satellite by taking a good look at the semantic view too.","metadata":{"papermill":{"duration":1.462623,"end_time":"2020-11-05T08:49:41.549332","exception":false,"start_time":"2020-11-05T08:49:40.086709","status":"completed"},"tags":[]}},{"cell_type":"code","source":"fig, axes = plt.subplots(1, 1, figsize=(15, 10))\nfor i, key in enumerate([\"py_satellite\", \"py_semantic\", 'stub_debug', \"box_debug\"]):\n    plot_image(key, ax=axes, agent=True)","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:49:45.266433Z","iopub.status.busy":"2020-11-05T08:49:45.265359Z","iopub.status.idle":"2020-11-05T08:52:39.093686Z","shell.execute_reply":"2020-11-05T08:52:39.094646Z"},"papermill":{"duration":176.314121,"end_time":"2020-11-05T08:52:39.094856","exception":false,"start_time":"2020-11-05T08:49:42.780735","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Again, box and stub will also give a good representaton of the data albeit with less low-level detail than the semantic view, seeing as the highways are not into much consideration here. The box view helps to just take a low-level look at the vehicles and their projected path whereas the stub view functions similarly to semantic. We can now proceed to taking a good look at the metadata provided by kkiller and potentially train a good model. The ones to check now will be stub and satellite to check.","metadata":{"papermill":{"duration":1.434874,"end_time":"2020-11-05T08:52:41.749841","exception":false,"start_time":"2020-11-05T08:52:40.314967","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Now as you can see with stub and box, under the hood the function uses the keys to create a rasterizer (for stub_debug for example StubRasterizer), which generates the AV surroundings and paths and passes it to an AgentDataset, with which we use to generate the predictions with our model. Also, there's a lot of meta-info about the rasterization that I want to have a look at, so let's look at metadata.","metadata":{"papermill":{"duration":1.564333,"end_time":"2020-11-05T08:52:44.470422","exception":false,"start_time":"2020-11-05T08:52:42.906089","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Metadata Exploration","metadata":{"papermill":{"duration":1.222823,"end_time":"2020-11-05T08:52:46.91982","exception":false,"start_time":"2020-11-05T08:52:45.696997","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Now that we can explore the images, we can also get a little down and dirty when it comes to the ZARR files. It's rather simple to use with the Python library for exploring them, especially the fact that it's NumPy interoperable.","metadata":{"papermill":{"duration":1.395443,"end_time":"2020-11-05T08:52:49.573566","exception":false,"start_time":"2020-11-05T08:52:48.178123","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(\"scenes\", zarr_dataset.scenes)\nprint(\"scenes[0]\", zarr_dataset.scenes[0])","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:52:52.332996Z","iopub.status.busy":"2020-11-05T08:52:52.330899Z","iopub.status.idle":"2020-11-05T08:52:52.335725Z","shell.execute_reply":"2020-11-05T08:52:52.333695Z"},"papermill":{"duration":1.536702,"end_time":"2020-11-05T08:52:52.335893","exception":false,"start_time":"2020-11-05T08:52:50.799191","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Also, a gentle note that we can use the ChunkedDataset to generate CSV files of scenes.","metadata":{"papermill":{"duration":1.151789,"end_time":"2020-11-05T08:52:54.678123","exception":false,"start_time":"2020-11-05T08:52:53.526334","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import pandas as pd\nscenes = zarr_dataset.scenes\nscenes_df = pd.DataFrame(scenes)\nscenes_df.columns = [\"data\"]; features = ['frame_index_interval', 'host', 'start_time', 'end_time']\nfor i, feature in enumerate(features):\n    scenes_df[feature] = scenes_df['data'].apply(lambda x: x[i])\nscenes_df.drop(columns=[\"data\"],inplace=True)\nprint(f\"scenes dataset: {scenes_df.shape}\")\nscenes_df.head()","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:52:57.05914Z","iopub.status.busy":"2020-11-05T08:52:57.058155Z","iopub.status.idle":"2020-11-05T08:52:57.081538Z","shell.execute_reply":"2020-11-05T08:52:57.080959Z"},"papermill":{"duration":1.185687,"end_time":"2020-11-05T08:52:57.08165","exception":false,"start_time":"2020-11-05T08:52:55.895963","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"But are we sure this is enough? Enough knowledge to satisfy us? No it's not. I will be using Kkiller's dataset for further tabular data exploration from henceforth.","metadata":{"papermill":{"duration":1.187891,"end_time":"2020-11-05T08:52:59.386253","exception":false,"start_time":"2020-11-05T08:52:58.198362","status":"completed"},"tags":[]}},{"cell_type":"code","source":"agents = pd.read_csv('../input/lyft-motion-prediction-autonomous-vehicles-as-csv/agents_0_10019001_10019001.csv')\nagents","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:53:01.89403Z","iopub.status.busy":"2020-11-05T08:53:01.893144Z","iopub.status.idle":"2020-11-05T08:53:29.401307Z","shell.execute_reply":"2020-11-05T08:53:29.400553Z"},"papermill":{"duration":28.80524,"end_time":"2020-11-05T08:53:29.401452","exception":false,"start_time":"2020-11-05T08:53:00.596212","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We have a literal wealth of information that we can use here to our benefits, including familiar features like:\n1. x, y,  and z coords\n2. yaw\n3. probabilites of other extraneous factors.","metadata":{"papermill":{"duration":1.205869,"end_time":"2020-11-05T08:53:31.802108","exception":false,"start_time":"2020-11-05T08:53:30.596239","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import seaborn as sns\ncolormap = plt.cm.magma\ncont_feats = [\"centroid_x\", \"centroid_y\", \"extent_x\", \"extent_y\", \"extent_z\", \"yaw\"]\nplt.figure(figsize=(16,12));\nplt.title('Pearson correlation of features', y=1.05, size=15);\nsns.heatmap(agents[cont_feats].corr(),linewidths=0.1,vmax=1.0, square=True, cmap=colormap, linecolor='white', annot=True);","metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-05T08:53:34.192435Z","iopub.status.busy":"2020-11-05T08:53:34.191763Z","iopub.status.idle":"2020-11-05T08:53:37.160427Z","shell.execute_reply":"2020-11-05T08:53:37.1613Z"},"papermill":{"duration":4.17449,"end_time":"2020-11-05T08:53:37.16151","exception":false,"start_time":"2020-11-05T08:53:32.98702","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Here we can extrapolate that the variables **centroid_x** and **centroid_y** have strongly negative correlations, and the strongest correlations are between **extent_z** and **extent_x** more than any other, coming in at 0.4. We can also try using an XGBoost/LightGBM model as kkiller has demonstrated in his brilliant kernel as an alternative approach to the problem.","metadata":{"papermill":{"duration":1.222117,"end_time":"2020-11-05T08:53:39.847456","exception":false,"start_time":"2020-11-05T08:53:38.625339","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### centroid_x and centroid_y","metadata":{"papermill":{"duration":1.364171,"end_time":"2020-11-05T08:53:42.442517","exception":false,"start_time":"2020-11-05T08:53:41.078346","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import seaborn as sns\nplot = sns.jointplot(x=agents['centroid_x'][:1000], y=agents['centroid_y'][:1000], kind='hexbin', color='blueviolet')\nplot.set_axis_labels('center_x', 'center_y', fontsize=16)\n\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:53:45.168919Z","iopub.status.busy":"2020-11-05T08:53:45.167841Z","iopub.status.idle":"2020-11-05T08:53:45.939585Z","shell.execute_reply":"2020-11-05T08:53:45.938927Z"},"papermill":{"duration":2.005408,"end_time":"2020-11-05T08:53:45.939725","exception":false,"start_time":"2020-11-05T08:53:43.934317","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"It seems like the two centroids have a somewhat strongly negative correlation and seemingly similar variable distributions. It seems that as such there is a negative correlation between both the variables.","metadata":{"papermill":{"duration":1.217239,"end_time":"2020-11-05T08:53:48.636821","exception":false,"start_time":"2020-11-05T08:53:47.419582","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### extent_x, extent_y and extent_z","metadata":{"papermill":{"duration":1.155285,"end_time":"2020-11-05T08:53:50.989313","exception":false,"start_time":"2020-11-05T08:53:49.834028","status":"completed"},"tags":[]}},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 15));\nsns.distplot(agents['extent_x'], color='steelblue');\nsns.distplot(agents['extent_y'], color='purple');\n\nplt.title(\"Distributions of Extents X and Y\");","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:53:53.329112Z","iopub.status.busy":"2020-11-05T08:53:53.327461Z","iopub.status.idle":"2020-11-05T08:54:03.922836Z","shell.execute_reply":"2020-11-05T08:54:03.923365Z"},"papermill":{"duration":11.776414,"end_time":"2020-11-05T08:54:03.923528","exception":false,"start_time":"2020-11-05T08:53:52.147114","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"It seems both the distributions of extent X and extent Y are heavily right skewed, as is centroid X. However, I have left out extent Z is order for readability of the plot, let's look at it now.\n\nTry to smooth the data and get:","metadata":{"papermill":{"duration":1.186868,"end_time":"2020-11-05T08:54:06.316256","exception":false,"start_time":"2020-11-05T08:54:05.129388","status":"completed"},"tags":[]}},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 15));\nsns.distplot(agents['extent_z'], color='steelblue');\n\nplt.title(\"Distributions of Extents z\");","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:54:09.543531Z","iopub.status.busy":"2020-11-05T08:54:09.542439Z","iopub.status.idle":"2020-11-05T08:54:14.892903Z","shell.execute_reply":"2020-11-05T08:54:14.891925Z"},"papermill":{"duration":7.398101,"end_time":"2020-11-05T08:54:14.893028","exception":false,"start_time":"2020-11-05T08:54:07.494927","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Once again, we have a right-skewed distribution as is the same with all the `extent` variables. ","metadata":{"papermill":{"duration":1.149401,"end_time":"2020-11-05T08:54:17.229404","exception":false,"start_time":"2020-11-05T08:54:16.080003","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### yaw","metadata":{"papermill":{"duration":1.420068,"end_time":"2020-11-05T08:54:19.809765","exception":false,"start_time":"2020-11-05T08:54:18.389697","status":"completed"},"tags":[]}},{"cell_type":"code","source":"fig = plt.figure(figsize=(15, 15));\nsns.distplot(agents['yaw'], color='steelblue');\n\nplt.title(\"Distributions of Extents z\");","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:54:22.209015Z","iopub.status.busy":"2020-11-05T08:54:22.207929Z","iopub.status.idle":"2020-11-05T08:54:27.477356Z","shell.execute_reply":"2020-11-05T08:54:27.476768Z"},"papermill":{"duration":6.500542,"end_time":"2020-11-05T08:54:27.477481","exception":false,"start_time":"2020-11-05T08:54:20.976939","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"So yes it seems like this distribution has several \"protrusions\" as I shall call them. We can now move on to exploring the frames data to check how feasible it is for our tabular purposes.","metadata":{"papermill":{"duration":1.18891,"end_time":"2020-11-05T08:54:29.873157","exception":false,"start_time":"2020-11-05T08:54:28.684247","status":"completed"},"tags":[]}},{"cell_type":"code","source":"frms = pd.read_csv(\"../input/lyft-motion-prediction-autonomous-vehicles-as-csv/frames_0_124167_124167.csv\")\nfrms.head()","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:54:32.572664Z","iopub.status.busy":"2020-11-05T08:54:32.571925Z","iopub.status.idle":"2020-11-05T08:54:33.117393Z","shell.execute_reply":"2020-11-05T08:54:33.116481Z"},"papermill":{"duration":1.73085,"end_time":"2020-11-05T08:54:33.117571","exception":false,"start_time":"2020-11-05T08:54:31.386721","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"So here we have the ego rotations with regards to the centroids, which will be very interesting to consider. It seems like we will require to check multiple of these variables at once:","metadata":{"papermill":{"duration":1.21298,"end_time":"2020-11-05T08:54:35.714594","exception":false,"start_time":"2020-11-05T08:54:34.501614","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### ego_rotatations","metadata":{"papermill":{"duration":1.595243,"end_time":"2020-11-05T08:54:38.4459","exception":false,"start_time":"2020-11-05T08:54:36.850657","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"First of all, we have nine ego rotation columns corresponding to each. So I would want to do a quick check of the correlation of these variables before moving on to some more high-level analyses.","metadata":{"papermill":{"duration":1.412976,"end_time":"2020-11-05T08:54:41.003906","exception":false,"start_time":"2020-11-05T08:54:39.59093","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import seaborn as sns\ncolormap = plt.cm.magma\ncont_feats = [\"ego_rotation_xx\", \"ego_rotation_xy\", \"ego_rotation_xz\", \"ego_rotation_yx\", \"ego_rotation_yy\", \"ego_rotation_yz\", \"ego_rotation_zx\", \"ego_rotation_zy\", \"ego_rotation_zz\"]\nplt.figure(figsize=(16,12));\nplt.title('Pearson correlation of features', y=1.05, size=15);\nsns.heatmap(frms[cont_feats].corr(),linewidths=0.1,vmax=1.0, square=True, \n            cmap=colormap, linecolor='white', annot=True);\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-05T08:54:44.608058Z","iopub.status.busy":"2020-11-05T08:54:44.60653Z","iopub.status.idle":"2020-11-05T08:54:45.26931Z","shell.execute_reply":"2020-11-05T08:54:45.269809Z"},"papermill":{"duration":2.243683,"end_time":"2020-11-05T08:54:45.269942","exception":false,"start_time":"2020-11-05T08:54:43.026259","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Things to note from this correlation analysis:\n1. The rotation coordinates with `y` and `z` seem to be uncorrelated most of the time\n2. The coordinates which have `x` are correlated strongly with the z-dimensional rotation (could this be indicative of something? I very much think so)","metadata":{"papermill":{"duration":1.179254,"end_time":"2020-11-05T08:54:47.64204","exception":false,"start_time":"2020-11-05T08:54:46.462786","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Augmentation ideas","metadata":{"papermill":{"duration":1.360154,"end_time":"2020-11-05T08:54:50.206625","exception":false,"start_time":"2020-11-05T08:54:48.846471","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"So the idea here is to augment the datas as discussed on the forum by @ryches (https://www.kaggle.com/c/lyft-motion-prediction-autonomous-vehicles/discussion/188368) since albumentations can use the data with keypoint format.","metadata":{"papermill":{"duration":1.471827,"end_time":"2020-11-05T08:54:52.898788","exception":false,"start_time":"2020-11-05T08:54:51.426961","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Plot with the cutout transformation:","metadata":{"papermill":{"duration":1.211608,"end_time":"2020-11-05T08:54:55.310143","exception":false,"start_time":"2020-11-05T08:54:54.098535","status":"completed"},"tags":[]}},{"cell_type":"code","source":"plot_with_tfms([A.Cutout()])","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:54:57.742968Z","iopub.status.busy":"2020-11-05T08:54:57.741614Z","iopub.status.idle":"2020-11-05T08:54:57.963144Z","shell.execute_reply":"2020-11-05T08:54:57.962204Z"},"papermill":{"duration":1.40757,"end_time":"2020-11-05T08:54:57.963273","exception":false,"start_time":"2020-11-05T08:54:56.555703","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Albumentations basically uses the keypoints provided in order to facilitate augmentation. (using cutout is dangerous because of loss of information possibilities btw, this is just for an example).","metadata":{"papermill":{"duration":1.173135,"end_time":"2020-11-05T08:55:00.338348","exception":false,"start_time":"2020-11-05T08:54:59.165213","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Let's see the pixel distributions for the above plot:","metadata":{"papermill":{"duration":1.196733,"end_time":"2020-11-05T08:55:02.752269","exception":false,"start_time":"2020-11-05T08:55:01.555536","status":"completed"},"tags":[]}},{"cell_type":"code","source":"tfms = A.Compose([A.Cutout()], keypoint_params=A.KeypointParams(format=\"xy\", remove_invisible=False))\nim = tfms(image=train_dataset_a[0]['image'], keypoints=train_dataset_a[0]['target_positions'])\nkeypoints = im['keypoints']\nim = train_dataset_a.rasterizer.to_rgb(im['image'].transpose(1, 2, 0))\nsns.distplot(im);","metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-05T08:55:05.358356Z","iopub.status.busy":"2020-11-05T08:55:05.357123Z","iopub.status.idle":"2020-11-05T08:55:05.698833Z","shell.execute_reply":"2020-11-05T08:55:05.69773Z"},"papermill":{"duration":1.512629,"end_time":"2020-11-05T08:55:05.698992","exception":false,"start_time":"2020-11-05T08:55:04.186363","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"And how much they have changed from the original:","metadata":{"papermill":{"duration":1.206429,"end_time":"2020-11-05T08:55:08.119811","exception":false,"start_time":"2020-11-05T08:55:06.913382","status":"completed"},"tags":[]}},{"cell_type":"code","source":"sns.distplot(train_dataset_a[0]['image']);","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:55:10.438907Z","iopub.status.busy":"2020-11-05T08:55:10.438013Z","iopub.status.idle":"2020-11-05T08:55:10.780418Z","shell.execute_reply":"2020-11-05T08:55:10.779787Z"},"papermill":{"duration":1.52613,"end_time":"2020-11-05T08:55:10.780561","exception":false,"start_time":"2020-11-05T08:55:09.254431","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Not much difference here except that we've shaved a lot off the `0.0` peak which is the effect of the cutout augmentatioin.","metadata":{"papermill":{"duration":1.143688,"end_time":"2020-11-05T08:55:13.112258","exception":false,"start_time":"2020-11-05T08:55:11.96857","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Baseline model (source: [here](https://github.com/lyft/l5kit/blob/master/examples/agent_motion_prediction/agent_motion_prediction.ipynb))","metadata":{"papermill":{"duration":1.188059,"end_time":"2020-11-05T08:55:16.269416","exception":false,"start_time":"2020-11-05T08:55:15.081357","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"![](https://raw.githubusercontent.com/catalyst-team/catalyst-pics/master/pics/catalyst_logo.png)","metadata":{"papermill":{"duration":1.177655,"end_time":"2020-11-05T08:55:18.651958","exception":false,"start_time":"2020-11-05T08:55:17.474303","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"This is mainly me using the Lyft baseline model and training it, for the purpose of demonstrating how we can fit to a PyTorch model with the provided dataset format. Also, I am using the tool neptune.ai for monitoring the epoch progress.\n\nAlso using wonderful ML library catalyst and hydra, which helps your PyTorch training greatly. For now, let's get started on the modelling with some basic import setup:","metadata":{"papermill":{"duration":1.14396,"end_time":"2020-11-05T08:55:20.947376","exception":false,"start_time":"2020-11-05T08:55:19.803416","status":"completed"},"tags":[]}},{"cell_type":"code","source":"cfg2 = load_config_data(\"../input/lyft-config-files/agent_motion_config.yaml\")\ncfg2 = omegaconf.OmegaConf.create(cfg2)\ntrain_cfg = omegaconf.OmegaConf.to_container(cfg2.train_data_loader)\nvalidation_cfg = omegaconf.OmegaConf.to_container(cfg2.val_data_loader)\n# Rasterizer\nrasterizer = build_rasterizer(cfg2, dm)","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:55:23.331589Z","iopub.status.busy":"2020-11-05T08:55:23.330391Z","iopub.status.idle":"2020-11-05T08:55:29.563Z","shell.execute_reply":"2020-11-05T08:55:29.563788Z"},"papermill":{"duration":7.443184,"end_time":"2020-11-05T08:55:29.563978","exception":false,"start_time":"2020-11-05T08:55:22.120794","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now we pretty-print the structure of our configuration. Hydra allows you to easily manage your configuration structure and schema whenever you train a machine learning model - it is also, as an additional benefit, part of the official **Pytorch Ecosystem.**","metadata":{"papermill":{"duration":1.168811,"end_time":"2020-11-05T08:55:31.910634","exception":false,"start_time":"2020-11-05T08:55:30.741823","status":"completed"},"tags":[]}},{"cell_type":"code","source":"print(cfg2.pretty())","metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-05T08:55:34.253846Z","iopub.status.busy":"2020-11-05T08:55:34.252364Z","iopub.status.idle":"2020-11-05T08:55:34.25736Z","shell.execute_reply":"2020-11-05T08:55:34.256437Z"},"papermill":{"duration":1.185878,"end_time":"2020-11-05T08:55:34.257525","exception":false,"start_time":"2020-11-05T08:55:33.071647","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This is basically initializing the rasterization process for the training data which basically functions to pass the .zarr files to our model and make it a coherent data format easily modular with our PyTorch setup.","metadata":{"papermill":{"duration":1.455043,"end_time":"2020-11-05T08:55:36.885544","exception":false,"start_time":"2020-11-05T08:55:35.430501","status":"completed"},"tags":[]}},{"cell_type":"code","source":"class LyftModel(torch.nn.Module):\n    def __init__(self, cfg: omegaconf.dictconfig.DictConfig):\n        super().__init__()\n        self.backbone = smp.FPN(encoder_name=\"resnext50_32x4d\", classes=1)\n        num_history_channels = (cfg[\"model_params\"][\"history_num_frames\"] + 1) * 2\n        num_in_channels = 3 + num_history_channels\n        self.backbone.encoder.conv1 = nn.Conv2d(\n            num_in_channels,\n             self.backbone.encoder.conv1.out_channels,\n            kernel_size= self.backbone.encoder.conv1.kernel_size,\n            stride= self.backbone.encoder.conv1.stride,\n            padding= self.backbone.encoder.conv1.padding,\n            bias=False,\n        ) \n        backbone_out_features = 14\n        num_targets = 2 * cfg[\"model_params\"][\"future_num_frames\"]\n        self.head = nn.Sequential(\n            nn.Dropout(0.2),\n            nn.Linear(in_features=14, out_features=4096),\n        )\n        self.backbone.segmentation_head = nn.Sequential(nn.Conv1d(56, 1, kernel_size=3, stride=2), nn.Dropout(0.2), nn.ReLU())\n        self.logit = nn.Linear(4096, out_features=num_targets)\n        self.logit_final = nn.Linear(128, 12)\n        self.num_preds = num_targets * 3\n    def forward(self, x):\n        x = self.backbone.encoder.conv1(x)\n        x = self.backbone.encoder.bn1(x)        \n        x = self.backbone.encoder.relu(x)\n        x = self.backbone.encoder.maxpool(x)        \n        x = self.backbone.encoder.layer1(x)\n        x = self.backbone.encoder.layer2(x)\n        x = self.backbone.encoder.layer3(x)\n        x = self.backbone.encoder.layer4(x)        \n        x = self.backbone.decoder.p5(x)\n        x = self.backbone.decoder.seg_blocks[0](x)\n        x = self.backbone.decoder.merge(x)\n        x = self.backbone.segmentation_head(x)\n        x = self.backbone.encoder.maxpool(x)\n        x = torch.flatten(x, 1)\n        x = self.head(x)\n        x = self.logit(x)   \n        x = x.permute(1, 0)\n        x = self.logit_final(x)\n        return x","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:55:39.647043Z","iopub.status.busy":"2020-11-05T08:55:39.646191Z","iopub.status.idle":"2020-11-05T08:55:39.650393Z","shell.execute_reply":"2020-11-05T08:55:39.64973Z"},"papermill":{"duration":1.180233,"end_time":"2020-11-05T08:55:39.650506","exception":false,"start_time":"2020-11-05T08:55:38.470273","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"~~Unhide the above cell if you want to see the model, it's basically Peter's original work. All I am doing here is modifying the training pipeline to be Catalyst-compatible.~~\n\n~~Update 17-09-2020: The model is roughly based on what kkiller has accomplished with PointNet and is basically me using a modified FPN network - to first segment (encode and decode) - and then use a simple MLP to classify. The model is roughly simple but it gives me pretty good results on loss, maybe something worth looking into?~~\n\nThe model is not eactly meant to segment + classify, that was my mistake in clarifying. It is simply a resnet50 + CBR block (Conv - BatchNorm - ReLU) + final part to get preds.","metadata":{"papermill":{"duration":1.411397,"end_time":"2020-11-05T08:55:42.590863","exception":false,"start_time":"2020-11-05T08:55:41.179466","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Part 1: Catalyst training","metadata":{"papermill":{"duration":1.195374,"end_time":"2020-11-05T08:55:45.026383","exception":false,"start_time":"2020-11-05T08:55:43.831009","status":"completed"},"tags":[]}},{"cell_type":"code","source":"device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nmodel = LyftModel(cfg2)\nmodel.to(device)\ntrain_zarr = ChunkedDataset(dm.require(train_cfg['key'])).open()\ntrain_dataset = AgentDataset(cfg2, train_zarr, rasterizer)\ndel train_cfg['key']\nsubset = torch.utils.data.Subset(train_dataset, range(0, 1100))\ntrain_dataloader = DataLoader(subset,\n                              **train_cfg)\nval_zarr = ChunkedDataset(dm.require(validation_cfg['key'])).open()\ndel validation_cfg['key']\n\nval_dataset = AgentDataset(cfg2, val_zarr, rasterizer)\nsubset = torch.utils.data.Subset(val_dataset, range(0, 50))\nval_dataloader = DataLoader(subset,\n                              **validation_cfg)","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:55:48.313738Z","iopub.status.busy":"2020-11-05T08:55:48.311498Z","iopub.status.idle":"2020-11-05T08:55:56.620051Z","shell.execute_reply":"2020-11-05T08:55:56.619356Z"},"papermill":{"duration":9.842932,"end_time":"2020-11-05T08:55:56.62019","exception":false,"start_time":"2020-11-05T08:55:46.777258","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This takes a very small subset of the data that we have here (mainly because this training pipeline is purely for demonstration purposes with regards to Catalyst).","metadata":{"papermill":{"duration":1.478434,"end_time":"2020-11-05T08:55:59.275416","exception":false,"start_time":"2020-11-05T08:55:57.796982","status":"completed"},"tags":[]}},{"cell_type":"code","source":"optimizer = torch.optim.Adam(model.parameters(), lr=0.02)\n\nloaders = {\n    \"train\": train_dataloader,\n    \"valid\": val_dataloader\n}\n\nclass LyftRunner(dl.SupervisedRunner):\n    def predict_batch(self, batch):\n        return self.model(batch[0].to(self.device).view(batch[0].size(0), -1))\n    def _handle_batch(self, batch):\n        x, y = batch['image'], batch['target_positions']\n        y_hat = self.model(x).view(y.shape)\n        target_availabilities = batch[\"target_availabilities\"].unsqueeze(-1)\n        criterion = torch.nn.MSELoss(reduction=\"none\")\n        loss = criterion(y_hat, y)\n        loss = loss * target_availabilities\n        loss = loss.mean()\n        self.batch_metrics.update(\n            {\"loss\": loss}\n        )\n","metadata":{"execution":{"iopub.execute_input":"2020-11-05T08:56:01.741998Z","iopub.status.busy":"2020-11-05T08:56:01.741093Z","iopub.status.idle":"2020-11-05T08:56:01.745189Z","shell.execute_reply":"2020-11-05T08:56:01.744575Z"},"papermill":{"duration":1.217618,"end_time":"2020-11-05T08:56:01.745323","exception":false,"start_time":"2020-11-05T08:56:00.527705","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Few things:\n+ Configuration of the data loaders\n+ Beginning the Catalyst process\n+ Neptune, wandb  initialization","metadata":{"papermill":{"duration":1.245991,"end_time":"2020-11-05T08:56:04.324425","exception":false,"start_time":"2020-11-05T08:56:03.078434","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%time\ndevice = utils.get_device()\nrunner = LyftRunner(device=device, input_key=\"image\", input_target_key=\"target_positions\", output_key=\"logits\")\nif MONITORING:\n\n    runner.train(\n        model=model,\n        optimizer=optimizer,\n        loaders=loaders,\n        logdir=\"../working\",\n        num_epochs=4,\n        verbose=True,\n        load_best_on_end=True,\n        callbacks=[neptune_logger, BatchOverfitCallback(train=10, valid=0.5), \n                  EarlyStoppingCallback(\n            patience=2,\n            metric=\"loss\",\n            minimize=True,\n        ), WandbLogger(project=\"dertaismus\",name= 'Example')\n                  ]\n    )\nelse:\n    runner.train(\n        model=model,\n        optimizer=optimizer,\n        loaders=loaders,\n        logdir=\"../working\",\n        num_epochs=4,\n        verbose=True,\n        load_best_on_end=True,\n        callbacks=[BatchOverfitCallback(train=10, valid=0.5), \n                  EarlyStoppingCallback(\n            patience=2,\n            metric=\"loss\",\n            minimize=True,\n        )\n                  ]\n    )\n    \n# train for more steps and loss will not be low\n# or at least not as pathetic as the loss over here    ","metadata":{"_kg_hide-input":true,"execution":{"iopub.execute_input":"2020-11-05T08:56:06.704697Z","iopub.status.busy":"2020-11-05T08:56:06.703885Z","iopub.status.idle":"2020-11-05T09:04:16.930285Z","shell.execute_reply":"2020-11-05T09:04:16.929561Z"},"papermill":{"duration":491.37537,"end_time":"2020-11-05T09:04:16.930428","exception":false,"start_time":"2020-11-05T08:56:05.555058","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"And this is how the Neptune metrics look like:\n![](https://i.imgur.com/uspP0q0.png)","metadata":{"papermill":{"duration":1.407578,"end_time":"2020-11-05T09:04:19.782565","exception":false,"start_time":"2020-11-05T09:04:18.374987","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"To properly work with Neptune callbacks, replace the `U.` fields in the callback class with your API token, project name etc. Then, add it as a callback to Catalyst and watch the magic happen.","metadata":{"papermill":{"duration":1.39587,"end_time":"2020-11-05T09:04:22.573878","exception":false,"start_time":"2020-11-05T09:04:21.178008","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Part 2: Kekas training","metadata":{"papermill":{"duration":1.698452,"end_time":"2020-11-05T09:04:25.872813","exception":false,"start_time":"2020-11-05T09:04:24.174361","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Kekas is a more simpler Fastai-esque way to train your model. Apart from having an extremely brilliant name, kekas gives you the simplicity of fastai and allows you to easily train and infer models.","metadata":{"papermill":{"duration":1.39004,"end_time":"2020-11-05T09:04:28.645657","exception":false,"start_time":"2020-11-05T09:04:27.255617","status":"completed"},"tags":[]}},{"cell_type":"code","source":"dataowner = DataOwner(train_dataloader, val_dataloader, None)\ndef step_fn(model: torch.nn.Module,\n            batch: torch.Tensor) -> torch.Tensor:\n    \n    inp,t = batch[\"image\"], batch[\"target_positions\"]\n    model.logit_final = nn.Linear(128, t.shape[0]).cuda()\n    return model(inp).reshape(t.shape)","metadata":{"execution":{"iopub.execute_input":"2020-11-05T09:04:31.6664Z","iopub.status.busy":"2020-11-05T09:04:31.665288Z","iopub.status.idle":"2020-11-05T09:04:31.676Z","shell.execute_reply":"2020-11-05T09:04:31.67877Z"},"papermill":{"duration":1.449648,"end_time":"2020-11-05T09:04:31.679122","exception":false,"start_time":"2020-11-05T09:04:30.229474","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Keker is the equivalent of a Catalyst Runner.","metadata":{"papermill":{"duration":1.379396,"end_time":"2020-11-05T09:04:34.856156","exception":false,"start_time":"2020-11-05T09:04:33.47676","status":"completed"},"tags":[]}},{"cell_type":"code","source":"keker = Keker(model=model,\n              dataowner=dataowner,\n              criterion=torch.nn.MSELoss(),\n              step_fn=step_fn,\n              target_key=\"target_positions\",\n              opt=torch.optim.SGD,\n              opt_params={\"momentum\": 0.99})","metadata":{"execution":{"iopub.execute_input":"2020-11-05T09:04:38.259877Z","iopub.status.busy":"2020-11-05T09:04:38.259129Z","iopub.status.idle":"2020-11-05T09:04:38.270152Z","shell.execute_reply":"2020-11-05T09:04:38.269587Z"},"papermill":{"duration":1.772069,"end_time":"2020-11-05T09:04:38.270287","exception":false,"start_time":"2020-11-05T09:04:36.498218","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Similar to fastai - unfreeze model and then freeze to.","metadata":{"papermill":{"duration":1.821325,"end_time":"2020-11-05T09:04:41.501676","exception":false,"start_time":"2020-11-05T09:04:39.680351","status":"completed"},"tags":[]}},{"cell_type":"code","source":"keker.unfreeze(model_attr=\"backbone\")\n\nlayer_num = -1\nkeker.freeze_to(layer_num, model_attr=\"backbone\")","metadata":{"execution":{"iopub.execute_input":"2020-11-05T09:04:44.44996Z","iopub.status.busy":"2020-11-05T09:04:44.442001Z","iopub.status.idle":"2020-11-05T09:04:44.452063Z","shell.execute_reply":"2020-11-05T09:04:44.452751Z"},"papermill":{"duration":1.463688,"end_time":"2020-11-05T09:04:44.452919","exception":false,"start_time":"2020-11-05T09:04:42.989231","status":"completed"},"tags":[]},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"And now we `kek` (train) our model!","metadata":{"papermill":{"duration":1.69924,"end_time":"2020-11-05T09:04:47.719738","exception":false,"start_time":"2020-11-05T09:04:46.020498","status":"completed"},"tags":[]}},{"cell_type":"code","source":"keker.kek( lr=1e-3, \n            epochs=5,                  \n            logdir='train_logs')\n\n# loss computation is slightly faulty I'll try to fix it if possible\n# issue is that we need to multiply loss by target availabilities like we did in Catalyst model\n# however that does not work out of the box with kekas as far as I know, this is the main reason loss is going yoyo\n\nkeker.plot_kek('train_logs')","metadata":{"execution":{"iopub.execute_input":"2020-11-05T09:04:50.47699Z","iopub.status.busy":"2020-11-05T09:04:50.475956Z","iopub.status.idle":"2020-11-05T09:07:57.119322Z","shell.execute_reply":"2020-11-05T09:07:57.119887Z"},"papermill":{"duration":188.029624,"end_time":"2020-11-05T09:07:57.12007","exception":false,"start_time":"2020-11-05T09:04:49.090446","status":"completed"},"tags":[]},"outputs":[],"execution_count":null}]}