{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Special thanks to https://www.kaggle.com/rishabhiitbhu/eda-understanding-the-dataset-with-3d-plots","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n# !pip install --upgrade kaggle\n# %matplotlib inline\n!pip install -U lyft_dataset_sdk\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow as tf\nimport numpy as np\nimport seaborn as sns\nimport cv2\n\nfrom pathlib import Path\n# Load the SDK\nfrom lyft_dataset_sdk.lyftdataset import LyftDataset, MapMask\nfrom lyft_dataset_sdk.utils.data_classes import LidarPointCloud\nfrom PIL import Image, ImageDraw, ImageFont\nfrom IPython.display import display\nfrom seaborn import color_palette\nsns.set(color_codes=True)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-08-03T06:23:46.552413Z","iopub.execute_input":"2022-08-03T06:23:46.553178Z","iopub.status.idle":"2022-08-03T06:23:54.531092Z","shell.execute_reply.started":"2022-08-03T06:23:46.55282Z","shell.execute_reply":"2022-08-03T06:23:54.530018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lyft SDK requires symbolic links to folders that contain images, maps, lidar, and data\n!ln -s /kaggle/input/3d-object-detection-for-autonomous-vehicles/train_images images\n!ln -s /kaggle/input/3d-object-detection-for-autonomous-vehicles/train_maps maps\n!ln -s /kaggle/input/3d-object-detection-for-autonomous-vehicles/train_lidar lidar\n!ln -s /kaggle/input/3d-object-detection-for-autonomous-vehicles/train_data data","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T06:23:54.533892Z","iopub.execute_input":"2022-08-03T06:23:54.534262Z","iopub.status.idle":"2022-08-03T06:23:59.383882Z","shell.execute_reply.started":"2022-08-03T06:23:54.534209Z","shell.execute_reply":"2022-08-03T06:23:59.382724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ld = LyftDataset(data_path=\".\", json_path=\"data\")\n#start with test\nsample_data = ld.sample[0]\n\n\n# Get the meta data for the lidar data\nlidar_top = ld.get('sample_data', sample_data[\"data\"][\"LIDAR_TOP\"])\nlidar_top","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T06:23:59.385884Z","iopub.execute_input":"2022-08-03T06:23:59.386264Z","iopub.status.idle":"2022-08-03T06:24:23.218279Z","shell.execute_reply.started":"2022-08-03T06:23:59.386208Z","shell.execute_reply":"2022-08-03T06:24:23.217002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explore SDK Tool","metadata":{}},{"cell_type":"code","source":"dir(ld)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:23.220122Z","iopub.execute_input":"2022-08-03T06:24:23.220565Z","iopub.status.idle":"2022-08-03T06:24:23.229275Z","shell.execute_reply.started":"2022-08-03T06:24:23.220481Z","shell.execute_reply":"2022-08-03T06:24:23.228143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(ld.category)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:23.233697Z","iopub.execute_input":"2022-08-03T06:24:23.234533Z","iopub.status.idle":"2022-08-03T06:24:23.242931Z","shell.execute_reply.started":"2022-08-03T06:24:23.234459Z","shell.execute_reply":"2022-08-03T06:24:23.241551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ld.category","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:23.247919Z","iopub.execute_input":"2022-08-03T06:24:23.248316Z","iopub.status.idle":"2022-08-03T06:24:23.257059Z","shell.execute_reply.started":"2022-08-03T06:24:23.248256Z","shell.execute_reply":"2022-08-03T06:24:23.255801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ld.get('category', 'c4e92c2b854fb03382838643c36d01a349f5624a11f184042d91d916d19acdbd')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:23.258721Z","iopub.execute_input":"2022-08-03T06:24:23.259371Z","iopub.status.idle":"2022-08-03T06:24:23.273287Z","shell.execute_reply.started":"2022-08-03T06:24:23.259287Z","shell.execute_reply":"2022-08-03T06:24:23.27114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ld.sample_annotation[:2]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:23.275544Z","iopub.execute_input":"2022-08-03T06:24:23.276012Z","iopub.status.idle":"2022-08-03T06:24:23.287909Z","shell.execute_reply.started":"2022-08-03T06:24:23.275918Z","shell.execute_reply":"2022-08-03T06:24:23.287036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'# of Annotations: {}'.format(len(ld.sample_annotation))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T06:24:23.289479Z","iopub.execute_input":"2022-08-03T06:24:23.290096Z","iopub.status.idle":"2022-08-03T06:24:23.303123Z","shell.execute_reply.started":"2022-08-03T06:24:23.290025Z","shell.execute_reply":"2022-08-03T06:24:23.30187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ld.get('attribute', ld.sample_annotation[3]['attribute_tokens'][0])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:23.305076Z","iopub.execute_input":"2022-08-03T06:24:23.306252Z","iopub.status.idle":"2022-08-03T06:24:23.315891Z","shell.execute_reply.started":"2022-08-03T06:24:23.306179Z","shell.execute_reply":"2022-08-03T06:24:23.315092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Scenes\nThese are some scenes in the dataset.\n","metadata":{}},{"cell_type":"code","source":"ld.scene[:2]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:23.317627Z","iopub.execute_input":"2022-08-03T06:24:23.318358Z","iopub.status.idle":"2022-08-03T06:24:23.331768Z","shell.execute_reply.started":"2022-08-03T06:24:23.318293Z","shell.execute_reply":"2022-08-03T06:24:23.330501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'# of Scenes: {}'.format(len(ld.scene))","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-08-03T06:24:23.33572Z","iopub.execute_input":"2022-08-03T06:24:23.336097Z","iopub.status.idle":"2022-08-03T06:24:23.348328Z","shell.execute_reply.started":"2022-08-03T06:24:23.336031Z","shell.execute_reply":"2022-08-03T06:24:23.346779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('../input/3d-object-detection-for-autonomous-vehicles/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:23.350266Z","iopub.execute_input":"2022-08-03T06:24:23.351069Z","iopub.status.idle":"2022-08-03T06:24:24.062771Z","shell.execute_reply.started":"2022-08-03T06:24:23.351Z","shell.execute_reply":"2022-08-03T06:24:24.061816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:24.064635Z","iopub.execute_input":"2022-08-03T06:24:24.065274Z","iopub.status.idle":"2022-08-03T06:24:24.189046Z","shell.execute_reply.started":"2022-08-03T06:24:24.065208Z","shell.execute_reply":"2022-08-03T06:24:24.187773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:24.190461Z","iopub.execute_input":"2022-08-03T06:24:24.190753Z","iopub.status.idle":"2022-08-03T06:24:24.201683Z","shell.execute_reply.started":"2022-08-03T06:24:24.190702Z","shell.execute_reply":"2022-08-03T06:24:24.200496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"token = train_df.iloc[0]['Id']\ntoken","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:24.203534Z","iopub.execute_input":"2022-08-03T06:24:24.204336Z","iopub.status.idle":"2022-08-03T06:24:24.217626Z","shell.execute_reply.started":"2022-08-03T06:24:24.203916Z","shell.execute_reply":"2022-08-03T06:24:24.216478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sample\nScene keyframe. There are many samples (126 ish) in a given scene.","metadata":{}},{"cell_type":"code","source":"sample = ld.get('sample', token)\nsample","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:24.219098Z","iopub.execute_input":"2022-08-03T06:24:24.219687Z","iopub.status.idle":"2022-08-03T06:24:24.236394Z","shell.execute_reply.started":"2022-08-03T06:24:24.219625Z","shell.execute_reply":"2022-08-03T06:24:24.235639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Render 3d Sample\nUse SDK to render scene.","metadata":{}},{"cell_type":"code","source":"ld.render_sample_3d_interactive(sample['token'], render_sample=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:24.237596Z","iopub.execute_input":"2022-08-03T06:24:24.238042Z","iopub.status.idle":"2022-08-03T06:24:25.500798Z","shell.execute_reply.started":"2022-08-03T06:24:24.237987Z","shell.execute_reply":"2022-08-03T06:24:25.499821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ld.sensor","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:25.502194Z","iopub.execute_input":"2022-08-03T06:24:25.502657Z","iopub.status.idle":"2022-08-03T06:24:25.50903Z","shell.execute_reply.started":"2022-08-03T06:24:25.502612Z","shell.execute_reply":"2022-08-03T06:24:25.508267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check out camera data\nsens = 'CAM_BACK_RIGHT'\ncamera = ld.get('sample_data', sample['data'][sens])\ncamera","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:25.510221Z","iopub.execute_input":"2022-08-03T06:24:25.510621Z","iopub.status.idle":"2022-08-03T06:24:25.526392Z","shell.execute_reply.started":"2022-08-03T06:24:25.510578Z","shell.execute_reply":"2022-08-03T06:24:25.525319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = Image.open(camera['filename'])\nimg","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:25.528028Z","iopub.execute_input":"2022-08-03T06:24:25.528525Z","iopub.status.idle":"2022-08-03T06:24:26.161527Z","shell.execute_reply.started":"2022-08-03T06:24:25.528466Z","shell.execute_reply":"2022-08-03T06:24:26.160407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ld.render_sample_data(camera['token'], with_anns=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:26.162899Z","iopub.execute_input":"2022-08-03T06:24:26.163203Z","iopub.status.idle":"2022-08-03T06:24:26.621198Z","shell.execute_reply.started":"2022-08-03T06:24:26.163153Z","shell.execute_reply":"2022-08-03T06:24:26.620221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lidar = ld.get('sample_data',sample['data']['LIDAR_TOP'])\nlidar","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:26.622884Z","iopub.execute_input":"2022-08-03T06:24:26.62326Z","iopub.status.idle":"2022-08-03T06:24:26.630617Z","shell.execute_reply.started":"2022-08-03T06:24:26.623195Z","shell.execute_reply":"2022-08-03T06:24:26.6297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"points = LidarPointCloud.from_file(Path(lidar['filename']))\npoints.points.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:26.632461Z","iopub.execute_input":"2022-08-03T06:24:26.632849Z","iopub.status.idle":"2022-08-03T06:24:26.644991Z","shell.execute_reply.started":"2022-08-03T06:24:26.632785Z","shell.execute_reply":"2022-08-03T06:24:26.643126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ld.render_sample_data(lidar['token'], with_anns=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:26.647187Z","iopub.execute_input":"2022-08-03T06:24:26.647921Z","iopub.status.idle":"2022-08-03T06:24:31.389728Z","shell.execute_reply.started":"2022-08-03T06:24:26.647772Z","shell.execute_reply":"2022-08-03T06:24:31.388677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ld.get('calibrated_sensor', camera['calibrated_sensor_token'])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:31.391319Z","iopub.execute_input":"2022-08-03T06:24:31.391859Z","iopub.status.idle":"2022-08-03T06:24:31.400728Z","shell.execute_reply.started":"2022-08-03T06:24:31.391794Z","shell.execute_reply":"2022-08-03T06:24:31.399663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annot = ld.get('sample_annotation', sample['anns'][0])\nannot","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:31.402367Z","iopub.execute_input":"2022-08-03T06:24:31.402979Z","iopub.status.idle":"2022-08-03T06:24:31.413519Z","shell.execute_reply.started":"2022-08-03T06:24:31.402894Z","shell.execute_reply":"2022-08-03T06:24:31.41248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"box = ld.get_box(annot['token'])\nbox","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:31.414791Z","iopub.execute_input":"2022-08-03T06:24:31.415282Z","iopub.status.idle":"2022-08-03T06:24:31.425733Z","shell.execute_reply.started":"2022-08-03T06:24:31.415229Z","shell.execute_reply":"2022-08-03T06:24:31.424877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"box.center","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:31.426977Z","iopub.execute_input":"2022-08-03T06:24:31.427406Z","iopub.status.idle":"2022-08-03T06:24:31.438979Z","shell.execute_reply.started":"2022-08-03T06:24:31.42736Z","shell.execute_reply":"2022-08-03T06:24:31.437989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"box.wlh","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:31.440419Z","iopub.execute_input":"2022-08-03T06:24:31.441004Z","iopub.status.idle":"2022-08-03T06:24:31.448673Z","shell.execute_reply.started":"2022-08-03T06:24:31.440919Z","shell.execute_reply":"2022-08-03T06:24:31.447455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ld.render_annotation(annot['token'], margin=40)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:31.45059Z","iopub.execute_input":"2022-08-03T06:24:31.451265Z","iopub.status.idle":"2022-08-03T06:24:36.64756Z","shell.execute_reply.started":"2022-08-03T06:24:31.4512Z","shell.execute_reply":"2022-08-03T06:24:36.646648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Attributes\nCheck out these attributes","metadata":{}},{"cell_type":"code","source":"attr1 = ld.get('attribute', annot['attribute_tokens'][0])\nattr2 = ld.get('attribute', annot['attribute_tokens'][1])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:36.65346Z","iopub.execute_input":"2022-08-03T06:24:36.653868Z","iopub.status.idle":"2022-08-03T06:24:36.659303Z","shell.execute_reply.started":"2022-08-03T06:24:36.653799Z","shell.execute_reply":"2022-08-03T06:24:36.658009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"attr1","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:36.661172Z","iopub.execute_input":"2022-08-03T06:24:36.661636Z","iopub.status.idle":"2022-08-03T06:24:36.677831Z","shell.execute_reply.started":"2022-08-03T06:24:36.661576Z","shell.execute_reply":"2022-08-03T06:24:36.67708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"attr2","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:36.679601Z","iopub.execute_input":"2022-08-03T06:24:36.679889Z","iopub.status.idle":"2022-08-03T06:24:36.68922Z","shell.execute_reply.started":"2022-08-03T06:24:36.679849Z","shell.execute_reply":"2022-08-03T06:24:36.688336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"instance = ld.get('instance', annot['instance_token'])\ninstance","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:36.691102Z","iopub.execute_input":"2022-08-03T06:24:36.692077Z","iopub.status.idle":"2022-08-03T06:24:36.701533Z","shell.execute_reply.started":"2022-08-03T06:24:36.692009Z","shell.execute_reply":"2022-08-03T06:24:36.70029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ld.render_instance(instance['token'])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:36.703523Z","iopub.execute_input":"2022-08-03T06:24:36.704188Z","iopub.status.idle":"2022-08-03T06:24:41.601159Z","shell.execute_reply.started":"2022-08-03T06:24:36.704113Z","shell.execute_reply":"2022-08-03T06:24:41.600186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use first_annotation_token\nld.render_annotation(instance['first_annotation_token'])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:41.602707Z","iopub.execute_input":"2022-08-03T06:24:41.603054Z","iopub.status.idle":"2022-08-03T06:24:46.403503Z","shell.execute_reply.started":"2022-08-03T06:24:41.602977Z","shell.execute_reply":"2022-08-03T06:24:46.402458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Can also use last_annotation_token\nld.render_annotation(instance['last_annotation_token'])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:46.405189Z","iopub.execute_input":"2022-08-03T06:24:46.405537Z","iopub.status.idle":"2022-08-03T06:24:50.842612Z","shell.execute_reply.started":"2022-08-03T06:24:46.405472Z","shell.execute_reply":"2022-08-03T06:24:50.841393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ld.render_sample(token)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:24:50.843895Z","iopub.execute_input":"2022-08-03T06:24:50.84421Z","iopub.status.idle":"2022-08-03T06:25:10.888518Z","shell.execute_reply.started":"2022-08-03T06:24:50.844158Z","shell.execute_reply":"2022-08-03T06:25:10.887552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ld.list_scenes()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:10.890222Z","iopub.execute_input":"2022-08-03T06:25:10.890844Z","iopub.status.idle":"2022-08-03T06:25:10.982357Z","shell.execute_reply.started":"2022-08-03T06:25:10.890787Z","shell.execute_reply":"2022-08-03T06:25:10.981358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"first_scene = ld.scene[0]\nlast_scene = ld.scene[-1]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:10.983777Z","iopub.execute_input":"2022-08-03T06:25:10.984125Z","iopub.status.idle":"2022-08-03T06:25:10.990058Z","shell.execute_reply.started":"2022-08-03T06:25:10.984067Z","shell.execute_reply":"2022-08-03T06:25:10.989084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_token = first_scene[\"first_sample_token\"]\n# my_sample_token = level5data.get(\"sample\", my_sample_token)[\"next\"]  # proceed to next sample\n\nld.render_sample(sample_token)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:10.992068Z","iopub.execute_input":"2022-08-03T06:25:10.992462Z","iopub.status.idle":"2022-08-03T06:25:33.02547Z","shell.execute_reply.started":"2022-08-03T06:25:10.992392Z","shell.execute_reply":"2022-08-03T06:25:33.024249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_scene","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:33.02712Z","iopub.execute_input":"2022-08-03T06:25:33.027419Z","iopub.status.idle":"2022-08-03T06:25:33.033899Z","shell.execute_reply.started":"2022-08-03T06:25:33.027369Z","shell.execute_reply":"2022-08-03T06:25:33.03285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Last scene in the list\nlast_scene_token = last_scene[\"first_sample_token\"]\n# my_sample_token = level5data.get(\"sample\", my_sample_token)[\"next\"]  # proceed to next sample\n\nld.render_sample(last_scene_token)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:33.035452Z","iopub.execute_input":"2022-08-03T06:25:33.035761Z","iopub.status.idle":"2022-08-03T06:25:44.344584Z","shell.execute_reply.started":"2022-08-03T06:25:33.035703Z","shell.execute_reply":"2022-08-03T06:25:44.343574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_token = last_scene[\"last_sample_token\"]\nld.render_sample(last_token)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:44.346181Z","iopub.execute_input":"2022-08-03T06:25:44.346608Z","iopub.status.idle":"2022-08-03T06:25:53.492843Z","shell.execute_reply.started":"2022-08-03T06:25:44.346418Z","shell.execute_reply":"2022-08-03T06:25:53.491563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Model Hyperparameters \n_BATCH_NORM_DECAY = 0.9\n_BATCH_NORM_EPSILON = 1e-05\n_LEAKY_RELU = 0.1\n_ANCHORS = [(10, 13), (16, 30), (33, 23),\n            (30, 61), (62, 45), (59, 119),\n            (116, 90), (156, 198), (373, 326)]\n_MODEL_SIZE = (416, 416)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:53.494875Z","iopub.execute_input":"2022-08-03T06:25:53.495255Z","iopub.status.idle":"2022-08-03T06:25:53.502313Z","shell.execute_reply.started":"2022-08-03T06:25:53.49519Z","shell.execute_reply":"2022-08-03T06:25:53.501267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It's useful to define `batch_norm` function since the model uses batch norms with shared parameters heavily. Also, same as ResNet, Yolo uses convolution with fixed padding, which means that padding is defined only by the size of the kernel.","metadata":{}},{"cell_type":"code","source":"def batch_norm(inputs, training, data_format):\n    \"\"\"Performs a batch normalization using a standard set of parameters.\"\"\"\n    return tf.layers.batch_normalization(\n        inputs=inputs, axis=1 if data_format == 'channels_first' else 3,\n        momentum=_BATCH_NORM_DECAY, epsilon=_BATCH_NORM_EPSILON,\n        scale=True, training=training)\n\n\ndef fixed_padding(inputs, kernel_size, data_format):\n    \"\"\"ResNet implementation of fixed padding.\n\n    Pads the input along the spatial dimensions independently of input size.\n\n    Args:\n        inputs: Tensor input to be padded.\n        kernel_size: The kernel to be used in the conv2d or max_pool2d.\n        data_format: The input format.\n    Returns:\n        A tensor with the same format as the input.\n    \"\"\"\n    pad_total = kernel_size - 1\n    pad_beg = pad_total // 2\n    pad_end = pad_total - pad_beg\n\n    if data_format == 'channels_first':\n        padded_inputs = tf.pad(inputs, [[0, 0], [0, 0],\n                                        [pad_beg, pad_end],\n                                        [pad_beg, pad_end]])\n    else:\n        padded_inputs = tf.pad(inputs, [[0, 0], [pad_beg, pad_end],\n                                        [pad_beg, pad_end], [0, 0]])\n    return padded_inputs\n\n\ndef conv2d_fixed_padding(inputs, filters, kernel_size, data_format, strides=1):\n    \"\"\"Strided 2-D convolution with explicit padding.\"\"\"\n    if strides > 1:\n        inputs = fixed_padding(inputs, kernel_size, data_format)\n\n    return tf.layers.conv2d(\n        inputs=inputs, filters=filters, kernel_size=kernel_size,\n        strides=strides, padding=('SAME' if strides == 1 else 'VALID'),\n        use_bias=False, data_format=data_format)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:53.50377Z","iopub.execute_input":"2022-08-03T06:25:53.504092Z","iopub.status.idle":"2022-08-03T06:25:53.520691Z","shell.execute_reply.started":"2022-08-03T06:25:53.504034Z","shell.execute_reply":"2022-08-03T06:25:53.519239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"For feature extraction Yolo uses Darknet-53 neural net pretrained on ImageNet. Same as ResNet,  Darknet-53 has shortcut (residual) connections, which help information from earlier layers flow further. We omit the last 3 layers (Avgpool, Connected and Softmax) since we only need the features.","metadata":{}},{"cell_type":"code","source":"def darknet53_residual_block(inputs, filters, training, data_format,\n                             strides=1):\n    \"\"\"Creates a residual block for Darknet.\"\"\"\n    shortcut = inputs\n\n    inputs = conv2d_fixed_padding(\n        inputs, filters=filters, kernel_size=1, strides=strides,\n        data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n\n    inputs = conv2d_fixed_padding(\n        inputs, filters=2 * filters, kernel_size=3, strides=strides,\n        data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n\n    inputs += shortcut\n\n    return inputs\n\n\ndef darknet53(inputs, training, data_format):\n    \"\"\"Creates Darknet53 model for feature extraction.\"\"\"\n    inputs = conv2d_fixed_padding(inputs, filters=32, kernel_size=3,\n                                  data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n    inputs = conv2d_fixed_padding(inputs, filters=64, kernel_size=3,\n                                  strides=2, data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n\n    inputs = darknet53_residual_block(inputs, filters=32, training=training,\n                                      data_format=data_format)\n\n    inputs = conv2d_fixed_padding(inputs, filters=128, kernel_size=3,\n                                  strides=2, data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n\n    for _ in range(2):\n        inputs = darknet53_residual_block(inputs, filters=64,\n                                          training=training,\n                                          data_format=data_format)\n\n    inputs = conv2d_fixed_padding(inputs, filters=256, kernel_size=3,\n                                  strides=2, data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n\n    for _ in range(8):\n        inputs = darknet53_residual_block(inputs, filters=128,\n                                          training=training,\n                                          data_format=data_format)\n\n    route1 = inputs\n\n    inputs = conv2d_fixed_padding(inputs, filters=512, kernel_size=3,\n                                  strides=2, data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n\n    for _ in range(8):\n        inputs = darknet53_residual_block(inputs, filters=256,\n                                          training=training,\n                                          data_format=data_format)\n\n    route2 = inputs\n\n    inputs = conv2d_fixed_padding(inputs, filters=1024, kernel_size=3,\n                                  strides=2, data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n\n    for _ in range(4):\n        inputs = darknet53_residual_block(inputs, filters=512,\n                                          training=training,\n                                          data_format=data_format)\n\n    return route1, route2, inputs","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:53.523216Z","iopub.execute_input":"2022-08-03T06:25:53.523814Z","iopub.status.idle":"2022-08-03T06:25:53.545226Z","shell.execute_reply.started":"2022-08-03T06:25:53.523756Z","shell.execute_reply":"2022-08-03T06:25:53.54394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def yolo_convolution_block(inputs, filters, training, data_format):\n    \"\"\"Creates convolution operations layer used after Darknet.\"\"\"\n    inputs = conv2d_fixed_padding(inputs, filters=filters, kernel_size=1,\n                                  data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n\n    inputs = conv2d_fixed_padding(inputs, filters=2 * filters, kernel_size=3,\n                                  data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n\n    inputs = conv2d_fixed_padding(inputs, filters=filters, kernel_size=1,\n                                  data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n\n    inputs = conv2d_fixed_padding(inputs, filters=2 * filters, kernel_size=3,\n                                  data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n\n    inputs = conv2d_fixed_padding(inputs, filters=filters, kernel_size=1,\n                                  data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n\n    route = inputs\n\n    inputs = conv2d_fixed_padding(inputs, filters=2 * filters, kernel_size=3,\n                                  data_format=data_format)\n    inputs = batch_norm(inputs, training=training, data_format=data_format)\n    inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n\n    return route, inputs","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:53.546747Z","iopub.execute_input":"2022-08-03T06:25:53.547091Z","iopub.status.idle":"2022-08-03T06:25:53.564081Z","shell.execute_reply.started":"2022-08-03T06:25:53.547029Z","shell.execute_reply":"2022-08-03T06:25:53.56313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def yolo_layer(inputs, n_classes, anchors, img_size, data_format):\n    \"\"\"Creates Yolo final detection layer.\n\n    Detects boxes with respect to anchors.\n\n    Args:\n        inputs: Tensor input.\n        n_classes: Number of labels.\n        anchors: A list of anchor sizes.\n        img_size: The input size of the model.\n        data_format: The input format.\n\n    Returns:\n        Tensor output.\n    \"\"\"\n    n_anchors = len(anchors)\n\n    inputs = tf.layers.conv2d(inputs, filters=n_anchors * (5 + n_classes),\n                              kernel_size=1, strides=1, use_bias=True,\n                              data_format=data_format)\n\n    shape = inputs.get_shape().as_list()\n    grid_shape = shape[2:4] if data_format == 'channels_first' else shape[1:3]\n    if data_format == 'channels_first':\n        inputs = tf.transpose(inputs, [0, 2, 3, 1])\n    inputs = tf.reshape(inputs, [-1, n_anchors * grid_shape[0] * grid_shape[1],\n                                 5 + n_classes])\n\n    strides = (img_size[0] // grid_shape[0], img_size[1] // grid_shape[1])\n\n    box_centers, box_shapes, confidence, classes = \\\n        tf.split(inputs, [2, 2, 1, n_classes], axis=-1)\n\n    x = tf.range(grid_shape[0], dtype=tf.float32)\n    y = tf.range(grid_shape[1], dtype=tf.float32)\n    x_offset, y_offset = tf.meshgrid(x, y)\n    x_offset = tf.reshape(x_offset, (-1, 1))\n    y_offset = tf.reshape(y_offset, (-1, 1))\n    x_y_offset = tf.concat([x_offset, y_offset], axis=-1)\n    x_y_offset = tf.tile(x_y_offset, [1, n_anchors])\n    x_y_offset = tf.reshape(x_y_offset, [1, -1, 2])\n    box_centers = tf.nn.sigmoid(box_centers)\n    box_centers = (box_centers + x_y_offset) * strides\n\n    anchors = tf.tile(anchors, [grid_shape[0] * grid_shape[1], 1])\n    box_shapes = tf.exp(box_shapes) * tf.to_float(anchors)\n\n    confidence = tf.nn.sigmoid(confidence)\n\n    classes = tf.nn.sigmoid(classes)\n\n    inputs = tf.concat([box_centers, box_shapes,\n                        confidence, classes], axis=-1)\n\n    return inputs","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:53.565777Z","iopub.execute_input":"2022-08-03T06:25:53.566171Z","iopub.status.idle":"2022-08-03T06:25:53.583644Z","shell.execute_reply.started":"2022-08-03T06:25:53.566108Z","shell.execute_reply":"2022-08-03T06:25:53.582562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def upsample(inputs, out_shape, data_format):\n    \"\"\"Upsamples to `out_shape` using nearest neighbor interpolation.\"\"\"\n    if data_format == 'channels_first':\n        inputs = tf.transpose(inputs, [0, 2, 3, 1])\n        new_height = out_shape[3]\n        new_width = out_shape[2]\n    else:\n        new_height = out_shape[2]\n        new_width = out_shape[1]\n\n    inputs = tf.image.resize_nearest_neighbor(inputs, (new_height, new_width))\n\n    if data_format == 'channels_first':\n        inputs = tf.transpose(inputs, [0, 3, 1, 2])\n\n    return inputs","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:53.58635Z","iopub.execute_input":"2022-08-03T06:25:53.586731Z","iopub.status.idle":"2022-08-03T06:25:53.599104Z","shell.execute_reply.started":"2022-08-03T06:25:53.586631Z","shell.execute_reply":"2022-08-03T06:25:53.598224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_boxes(inputs):\n    \"\"\"Computes top left and bottom right points of the boxes.\"\"\"\n    center_x, center_y, width, height, confidence, classes = \\\n        tf.split(inputs, [1, 1, 1, 1, 1, -1], axis=-1)\n\n    top_left_x = center_x - width / 2\n    top_left_y = center_y - height / 2\n    bottom_right_x = center_x + width / 2\n    bottom_right_y = center_y + height / 2\n\n    boxes = tf.concat([top_left_x, top_left_y,\n                       bottom_right_x, bottom_right_y,\n                       confidence, classes], axis=-1)\n\n    return boxes\n\n\ndef non_max_suppression(inputs, n_classes, max_output_size, iou_threshold,\n                        confidence_threshold):\n    \"\"\"Performs non-max suppression separately for each class.\n\n    Args:\n        inputs: Tensor input.\n        n_classes: Number of classes.\n        max_output_size: Max number of boxes to be selected for each class.\n        iou_threshold: Threshold for the IOU.\n        confidence_threshold: Threshold for the confidence score.\n    Returns:\n        A list containing class-to-boxes dictionaries\n            for each sample in the batch.\n    \"\"\"\n    batch = tf.unstack(inputs)\n    boxes_dicts = []\n    for boxes in batch:\n        boxes = tf.boolean_mask(boxes, boxes[:, 4] > confidence_threshold)\n        classes = tf.argmax(boxes[:, 5:], axis=-1)\n        classes = tf.expand_dims(tf.to_float(classes), axis=-1)\n        boxes = tf.concat([boxes[:, :5], classes], axis=-1)\n\n        boxes_dict = dict()\n        for cls in range(n_classes):\n            mask = tf.equal(boxes[:, 5], cls)\n            mask_shape = mask.get_shape()\n            if mask_shape.ndims != 0:\n                class_boxes = tf.boolean_mask(boxes, mask)\n                boxes_coords, boxes_conf_scores, _ = tf.split(class_boxes,\n                                                              [4, 1, -1],\n                                                              axis=-1)\n                boxes_conf_scores = tf.reshape(boxes_conf_scores, [-1])\n                indices = tf.image.non_max_suppression(boxes_coords,\n                                                       boxes_conf_scores,\n                                                       max_output_size,\n                                                       iou_threshold)\n                class_boxes = tf.gather(class_boxes, indices)\n                boxes_dict[cls] = class_boxes[:, :5]\n\n        boxes_dicts.append(boxes_dict)\n\n    return boxes_dicts","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:53.600487Z","iopub.execute_input":"2022-08-03T06:25:53.601078Z","iopub.status.idle":"2022-08-03T06:25:53.617511Z","shell.execute_reply.started":"2022-08-03T06:25:53.601015Z","shell.execute_reply":"2022-08-03T06:25:53.6164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Yolo_v3:\n    \"\"\"Yolo v3 model class.\"\"\"\n\n    def __init__(self, n_classes, model_size, max_output_size, iou_threshold,\n                 confidence_threshold, data_format=None):\n        \"\"\"Creates the model.\n\n        Args:\n            n_classes: Number of class labels.\n            model_size: The input size of the model.\n            max_output_size: Max number of boxes to be selected for each class.\n            iou_threshold: Threshold for the IOU.\n            confidence_threshold: Threshold for the confidence score.\n            data_format: The input format.\n\n        Returns:\n            None.\n        \"\"\"\n        if not data_format:\n            if tf.test.is_built_with_cuda():\n                data_format = 'channels_first'\n            else:\n                data_format = 'channels_last'\n\n        self.n_classes = n_classes\n        self.model_size = model_size\n        self.max_output_size = max_output_size\n        self.iou_threshold = iou_threshold\n        self.confidence_threshold = confidence_threshold\n        self.data_format = data_format\n\n    def __call__(self, inputs, training):\n        \"\"\"Add operations to detect boxes for a batch of input images.\n\n        Args:\n            inputs: A Tensor representing a batch of input images.\n            training: A boolean, whether to use in training or inference mode.\n\n        Returns:\n            A list containing class-to-boxes dictionaries\n                for each sample in the batch.\n        \"\"\"\n        with tf.variable_scope('yolo_v3_model'):\n            if self.data_format == 'channels_first':\n                inputs = tf.transpose(inputs, [0, 3, 1, 2])\n\n            inputs = inputs / 255\n\n            route1, route2, inputs = darknet53(inputs, training=training,\n                                               data_format=self.data_format)\n\n            route, inputs = yolo_convolution_block(\n                inputs, filters=512, training=training,\n                data_format=self.data_format)\n            detect1 = yolo_layer(inputs, n_classes=self.n_classes,\n                                 anchors=_ANCHORS[6:9],\n                                 img_size=self.model_size,\n                                 data_format=self.data_format)\n\n            inputs = conv2d_fixed_padding(route, filters=256, kernel_size=1,\n                                          data_format=self.data_format)\n            inputs = batch_norm(inputs, training=training,\n                                data_format=self.data_format)\n            inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n            upsample_size = route2.get_shape().as_list()\n            inputs = upsample(inputs, out_shape=upsample_size,\n                              data_format=self.data_format)\n            axis = 1 if self.data_format == 'channels_first' else 3\n            inputs = tf.concat([inputs, route2], axis=axis)\n            route, inputs = yolo_convolution_block(\n                inputs, filters=256, training=training,\n                data_format=self.data_format)\n            detect2 = yolo_layer(inputs, n_classes=self.n_classes,\n                                 anchors=_ANCHORS[3:6],\n                                 img_size=self.model_size,\n                                 data_format=self.data_format)\n\n            inputs = conv2d_fixed_padding(route, filters=128, kernel_size=1,\n                                          data_format=self.data_format)\n            inputs = batch_norm(inputs, training=training,\n                                data_format=self.data_format)\n            inputs = tf.nn.leaky_relu(inputs, alpha=_LEAKY_RELU)\n            upsample_size = route1.get_shape().as_list()\n            inputs = upsample(inputs, out_shape=upsample_size,\n                              data_format=self.data_format)\n            inputs = tf.concat([inputs, route1], axis=axis)\n            route, inputs = yolo_convolution_block(\n                inputs, filters=128, training=training,\n                data_format=self.data_format)\n            detect3 = yolo_layer(inputs, n_classes=self.n_classes,\n                                 anchors=_ANCHORS[0:3],\n                                 img_size=self.model_size,\n                                 data_format=self.data_format)\n\n            inputs = tf.concat([detect1, detect2, detect3], axis=1)\n\n            inputs = build_boxes(inputs)\n\n            boxes_dicts = non_max_suppression(\n                inputs, n_classes=self.n_classes,\n                max_output_size=self.max_output_size,\n                iou_threshold=self.iou_threshold,\n                confidence_threshold=self.confidence_threshold)\n\n            return boxes_dicts","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:53.619563Z","iopub.execute_input":"2022-08-03T06:25:53.620234Z","iopub.status.idle":"2022-08-03T06:25:53.644022Z","shell.execute_reply.started":"2022-08-03T06:25:53.620173Z","shell.execute_reply":"2022-08-03T06:25:53.643162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_images(img_names, model_size):\n    \"\"\"Loads images in a 4D array.\n\n    Args:\n        img_names: A list of images names.\n        model_size: The input size of the model.\n        data_format: A format for the array returned\n            ('channels_first' or 'channels_last').\n\n    Returns:\n        A 4D NumPy array.\n    \"\"\"\n    imgs = []\n\n    for img_name in img_names:\n        img = Image.open(img_name)\n        img = img.resize(size=model_size)\n        img = np.array(img, dtype=np.float32)\n        img = np.expand_dims(img, axis=0)\n        imgs.append(img)\n\n    imgs = np.concatenate(imgs)\n\n    return imgs\n\n\ndef load_class_names(file_name):\n    \"\"\"Returns a list of class names read from `file_name`.\"\"\"\n    with open(file_name, 'r') as f:\n        class_names = f.read().splitlines()\n    return class_names\n\n\ndef draw_boxes(img_names, boxes_dicts, class_names, model_size):\n    \"\"\"Draws detected boxes.\n\n    Args:\n        img_names: A list of input images names.\n        boxes_dict: A class-to-boxes dictionary.\n        class_names: A class names list.\n        model_size: The input size of the model.\n\n    Returns:\n        None.\n    \"\"\"\n    colors = ((np.array(color_palette(\"hls\", 80)) * 255)).astype(np.uint8)\n    for num, img_name, boxes_dict in zip(range(len(img_names)), img_names,\n                                         boxes_dicts):\n        img = Image.open(img_name)\n        draw = ImageDraw.Draw(img)\n        font = ImageFont.truetype(font='../input/futur.ttf',\n                                  size=(img.size[0] + img.size[1]) // 100)\n        resize_factor = \\\n            (img.size[0] / model_size[0], img.size[1] / model_size[1])\n        for cls in range(len(class_names)):\n            boxes = boxes_dict[cls]\n            if np.size(boxes) != 0:\n                color = colors[cls]\n                for box in boxes:\n                    xy, confidence = box[:4], box[4]\n                    xy = [xy[i] * resize_factor[i % 2] for i in range(4)]\n                    x0, y0 = xy[0], xy[1]\n                    thickness = (img.size[0] + img.size[1]) // 200\n                    for t in np.linspace(0, 1, thickness):\n                        xy[0], xy[1] = xy[0] + t, xy[1] + t\n                        xy[2], xy[3] = xy[2] - t, xy[3] - t\n                        draw.rectangle(xy, outline=tuple(color))\n                    text = '{} {:.1f}%'.format(class_names[cls],\n                                               confidence * 100)\n                    text_size = draw.textsize(text, font=font)\n                    draw.rectangle(\n                        [x0, y0 - text_size[1], x0 + text_size[0], y0],\n                        fill=tuple(color))\n                    draw.text((x0, y0 - text_size[1]), text, fill='black',\n                              font=font)\n\n        display(img)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:53.645207Z","iopub.execute_input":"2022-08-03T06:25:53.645749Z","iopub.status.idle":"2022-08-03T06:25:53.665896Z","shell.execute_reply.started":"2022-08-03T06:25:53.6457Z","shell.execute_reply":"2022-08-03T06:25:53.664785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_weights(variables, file_name):\n    \"\"\"Reshapes and loads official pretrained Yolo weights.\n\n    Args:\n        variables: A list of tf.Variable to be assigned.\n        file_name: A name of a file containing weights.\n\n    Returns:\n        A list of assign operations.\n    \"\"\"\n    with open(file_name, \"rb\") as f:\n        # Skip first 5 values containing irrelevant info\n        np.fromfile(f, dtype=np.int32, count=5)\n        weights = np.fromfile(f, dtype=np.float32)\n\n        assign_ops = []\n        ptr = 0\n\n        # Load weights for Darknet part.\n        # Each convolution layer has batch normalization.\n        for i in range(52):\n            conv_var = variables[5 * i]\n            gamma, beta, mean, variance = variables[5 * i + 1:5 * i + 5]\n            batch_norm_vars = [beta, gamma, mean, variance]\n\n            for var in batch_norm_vars:\n                shape = var.shape.as_list()\n                num_params = np.prod(shape)\n                var_weights = weights[ptr:ptr + num_params].reshape(shape)\n                ptr += num_params\n                assign_ops.append(tf.assign(var, var_weights))\n\n            shape = conv_var.shape.as_list()\n            num_params = np.prod(shape)\n            var_weights = weights[ptr:ptr + num_params].reshape(\n                (shape[3], shape[2], shape[0], shape[1]))\n            var_weights = np.transpose(var_weights, (2, 3, 1, 0))\n            ptr += num_params\n            assign_ops.append(tf.assign(conv_var, var_weights))\n\n        # Loading weights for Yolo part.\n        # 7th, 15th and 23rd convolution layer has biases and no batch norm.\n        ranges = [range(0, 6), range(6, 13), range(13, 20)]\n        unnormalized = [6, 13, 20]\n        for j in range(3):\n            for i in ranges[j]:\n                current = 52 * 5 + 5 * i + j * 2\n                conv_var = variables[current]\n                gamma, beta, mean, variance =  \\\n                    variables[current + 1:current + 5]\n                batch_norm_vars = [beta, gamma, mean, variance]\n\n                for var in batch_norm_vars:\n                    shape = var.shape.as_list()\n                    num_params = np.prod(shape)\n                    var_weights = weights[ptr:ptr + num_params].reshape(shape)\n                    ptr += num_params\n                    assign_ops.append(tf.assign(var, var_weights))\n\n                shape = conv_var.shape.as_list()\n                num_params = np.prod(shape)\n                var_weights = weights[ptr:ptr + num_params].reshape(\n                    (shape[3], shape[2], shape[0], shape[1]))\n                var_weights = np.transpose(var_weights, (2, 3, 1, 0))\n                ptr += num_params\n                assign_ops.append(tf.assign(conv_var, var_weights))\n\n            bias = variables[52 * 5 + unnormalized[j] * 5 + j * 2 + 1]\n            shape = bias.shape.as_list()\n            num_params = np.prod(shape)\n            var_weights = weights[ptr:ptr + num_params].reshape(shape)\n            ptr += num_params\n            assign_ops.append(tf.assign(bias, var_weights))\n\n            conv_var = variables[52 * 5 + unnormalized[j] * 5 + j * 2]\n            shape = conv_var.shape.as_list()\n            num_params = np.prod(shape)\n            var_weights = weights[ptr:ptr + num_params].reshape(\n                (shape[3], shape[2], shape[0], shape[1]))\n            var_weights = np.transpose(var_weights, (2, 3, 1, 0))\n            ptr += num_params\n            assign_ops.append(tf.assign(conv_var, var_weights))\n\n    return assign_ops","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:53.667614Z","iopub.execute_input":"2022-08-03T06:25:53.667925Z","iopub.status.idle":"2022-08-03T06:25:53.692594Z","shell.execute_reply.started":"2022-08-03T06:25:53.667867Z","shell.execute_reply":"2022-08-03T06:25:53.691136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_names = ['../data/data-for-yolo-v3-kernel/dog.jpg', '../data/data-for-yolo-v3-kernel/office.jpg']\nfor img in img_names: display(Image.open(img))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:53.694316Z","iopub.execute_input":"2022-08-03T06:25:53.694917Z","iopub.status.idle":"2022-08-03T06:25:53.736933Z","shell.execute_reply.started":"2022-08-03T06:25:53.694848Z","shell.execute_reply":"2022-08-03T06:25:53.735349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd ","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:53.73836Z","iopub.status.idle":"2022-08-03T06:25:53.739156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = len(img_names)\nbatch = load_images(img_names, model_size=_MODEL_SIZE)\nclass_names = load_class_names('./data-for-yolo-v3-kernel/coco.names')\nn_classes = len(class_names)\nmax_output_size = 10\niou_threshold = 0.5\nconfidence_threshold = 0.5\n\nmodel = Yolo_v3(n_classes=n_classes, model_size=_MODEL_SIZE,\n                max_output_size=max_output_size,\n                iou_threshold=iou_threshold,\n                confidence_threshold=confidence_threshold)\n\ninputs = tf.placeholder(tf.float32, [batch_size, 416, 416, 3])\n\ndetections = model(inputs, training=False)\n\nmodel_vars = tf.global_variables(scope='yolo_v3_model')\nassign_ops = load_weights(model_vars, './data-for-yolo-v3-kernel/yolov3.weights')\n\nwith tf.Session() as sess:\n    sess.run(assign_ops)\n    detection_result = sess.run(detections, feed_dict={inputs: batch})\n    \ndraw_boxes(img_names, detection_result, class_names, _MODEL_SIZE)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:25:53.740624Z","iopub.status.idle":"2022-08-03T06:25:53.741386Z"},"trusted":true},"execution_count":null,"outputs":[]}]}