{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8172274,"sourceType":"datasetVersion","datasetId":4836781},{"sourceId":8172395,"sourceType":"datasetVersion","datasetId":4836896}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Summary\n\n* Compared the inference speed of the EfficientNetB0 (4x3x224x224) using PyTorch, TorchScript, ONNX, and OpenVINO.\n* Inference speed is OpenVINO > ONNX >>> TorchScript > PyTorch.\n* Packages such as OpenVINO and ONNXRuntime have been registered with a public dataset for offline installation.\n\n# Update\n\nver5: I improved the readability by collapsing the code.","metadata":{"execution":{"iopub.status.busy":"2024-04-20T11:45:40.612527Z","iopub.execute_input":"2024-04-20T11:45:40.614218Z","iopub.status.idle":"2024-04-20T11:45:40.627616Z","shell.execute_reply.started":"2024-04-20T11:45:40.614140Z","shell.execute_reply":"2024-04-20T11:45:40.624553Z"}}},{"cell_type":"markdown","source":"# Install libraries","metadata":{}},{"cell_type":"code","source":"# onnxsim-0.4.36\n!pip install --no-index /kaggle/input/birdclef2024-openvino-onnxruntime/onnxsim-0.4.36/onnxsim-0.4.36-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n\n# onnxruntime-1.17.3\n!pip install --no-index /kaggle/input/birdclef2024-openvino-onnxruntime/onnxruntime-1.17.3/humanfriendly-10.0-py2.py3-none-any.whl\n!pip install --no-index /kaggle/input/birdclef2024-openvino-onnxruntime/onnxruntime-1.17.3/coloredlogs-15.0.1-py2.py3-none-any.whl\n!pip install --no-index /kaggle/input/birdclef2024-openvino-onnxruntime/onnxruntime-1.17.3/onnxruntime-1.17.3-cp310-cp310-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl\n\n# onnxconverter-common-1.14.0\n!pip install --no-index /kaggle/input/birdclef2024-openvino-onnxruntime/onnxconverter-common-1.14.0/protobuf-3.20.2-cp310-cp310-manylinux_2_12_x86_64.manylinux2010_x86_64.whl\n!pip install --no-index /kaggle/input/birdclef2024-openvino-onnxruntime/onnxconverter-common-1.14.0/onnxconverter_common-1.14.0-py2.py3-none-any.whl\n!pip install --no-index /kaggle/input/birdclef2024-openvino-onnxruntime/fastjsonschema-2.17.1/fastjsonschema-2.17.1-py3-none-any.whl\n\n# openvino-dev-2024.0.0\n!pip install --no-index /kaggle/input/birdclef2024-openvino-onnxruntime/onnx-1.15.0/onnx-1.15.0-cp310-cp310-manylinux_2_17_x86_64.manylinux2014_x86_64.whl\n!pip install --no-index /kaggle/input/birdclef2024-openvino-onnxruntime/openvino-dev-2024.0.0/networkx-3.1-py3-none-any.whl\n!pip install --no-index /kaggle/input/birdclef2024-openvino-onnxruntime/openvino-dev-2024.0.0/openvino_telemetry-2024.1.0-py3-none-any.whl\n!pip install --no-index /kaggle/input/birdclef2024-openvino-onnxruntime/openvino-dev-2024.0.0/openvino-2024.0.0-14509-cp310-cp310-manylinux2014_x86_64.whl\n!pip install --no-index /kaggle/input/birdclef2024-openvino-onnxruntime/openvino-dev-2024.0.0/openvino_dev-2024.0.0-14509-py3-none-any.whl","metadata":{"execution":{"iopub.status.busy":"2024-04-20T11:21:35.388850Z","iopub.execute_input":"2024-04-20T11:21:35.389375Z","iopub.status.idle":"2024-04-20T11:22:23.428371Z","shell.execute_reply.started":"2024-04-20T11:21:35.389339Z","shell.execute_reply":"2024-04-20T11:22:23.426021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import","metadata":{}},{"cell_type":"code","source":"import time\nimport timeit\nfrom pathlib import Path\n\nimport numpy as np\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nimport timm\n\nimport onnx\nimport onnxruntime as rt\nfrom onnxconverter_common import float16\n\nimport openvino\nimport openvino as ov\n\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_style('whitegrid')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-20T13:16:12.764528Z","iopub.execute_input":"2024-04-20T13:16:12.765003Z","iopub.status.idle":"2024-04-20T13:16:12.775203Z","shell.execute_reply.started":"2024-04-20T13:16:12.764967Z","shell.execute_reply":"2024-04-20T13:16:12.773722Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config","metadata":{}},{"cell_type":"code","source":"class CFG:\n    REPEAT: int = 5\n    LOOP: int = 40\n    INPUT_SHAPE: list[int] = [4, 3, 224, 224]\n    DUMMY_INPUT_TENSOR: torch.Tensor = torch.randn(*INPUT_SHAPE)\n    DUMMY_INPUT_NUMPY_FP32: np.ndarray = DUMMY_INPUT_TENSOR.numpy()\n    DUMMY_INPUT_NUMPY_FP16: np.ndarray = DUMMY_INPUT_NUMPY_FP32.astype(np.float16)\n    OUTPUT_DIR_ONNX: Path = Path('./model/onnx')\n    OUTPUT_DIR_OV: Path = Path('./model/ov')","metadata":{"execution":{"iopub.status.busy":"2024-04-20T13:16:15.055349Z","iopub.execute_input":"2024-04-20T13:16:15.056657Z","iopub.status.idle":"2024-04-20T13:16:15.078571Z","shell.execute_reply.started":"2024-04-20T13:16:15.056610Z","shell.execute_reply":"2024-04-20T13:16:15.076627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CFG.OUTPUT_DIR_ONNX.mkdir(parents=True)\nCFG.OUTPUT_DIR_OV.mkdir(parents=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T12:38:03.523963Z","iopub.execute_input":"2024-04-20T12:38:03.524379Z","iopub.status.idle":"2024-04-20T12:38:03.748846Z","shell.execute_reply.started":"2024-04-20T12:38:03.524347Z","shell.execute_reply":"2024-04-20T12:38:03.747307Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"class GeM(nn.Module):\n    def __init__(self, h: int, w: int, p: int = 3, eps: float = 1e-6):\n        super(GeM, self).__init__()\n        self.p = torch.nn.Parameter(torch.ones(1) * p)\n        self.eps = eps\n        self.h = h\n        self.w = w\n\n    def forward(self, x: torch.Tensor) -> torch.Tensor:\n        return F.avg_pool2d(x.clamp(min=self.eps).pow(self.p), (self.h, self.w)).pow(1.0 / self.p)[:, :, 0, 0]\n\n    \nmodel = timm.create_model(\n    'tf_efficientnet_b0.ns_jft_in1k',\n    features_only=False,\n    pretrained=False,\n    in_chans=3,\n)\n\nmodel.load_state_dict(torch.load('/kaggle/input/timm-tfefficientnetb0ns-pretrained/tf_efficientnet_b0_ns-c0e6a31c.pth'))\nmodel.global_pool = GeM(h=7, w=7)\nmodel.classifier = nn.Sequential(\n    nn.Linear(model.num_features, 182),\n    nn.Sigmoid(),\n)\nmodel.eval()","metadata":{"execution":{"iopub.status.busy":"2024-04-20T13:16:46.729709Z","iopub.execute_input":"2024-04-20T13:16:46.730173Z","iopub.status.idle":"2024-04-20T13:16:47.007694Z","shell.execute_reply.started":"2024-04-20T13:16:46.730138Z","shell.execute_reply":"2024-04-20T13:16:47.006540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reference_tensor = model(CFG.DUMMY_INPUT_TENSOR)\nreference_numpy = reference_tensor.detach().numpy()\n[model(CFG.DUMMY_INPUT_TENSOR) for i in range(5)]\nwith torch.no_grad():\n    pytorch_fp32_time_ms = timeit.timeit(lambda: model(CFG.DUMMY_INPUT_TENSOR), number=CFG.LOOP) / CFG.LOOP * 1000\nprint(f'time: {pytorch_fp32_time_ms:.2f} msec (n={CFG.LOOP})')","metadata":{"execution":{"iopub.status.busy":"2024-04-20T13:26:18.704667Z","iopub.execute_input":"2024-04-20T13:26:18.705225Z","iopub.status.idle":"2024-04-20T13:26:25.465023Z","shell.execute_reply.started":"2024-04-20T13:26:18.705163Z","shell.execute_reply":"2024-04-20T13:26:25.463806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert","metadata":{}},{"cell_type":"markdown","source":"## TorchScript","metadata":{"execution":{"iopub.status.busy":"2024-04-20T13:25:23.332638Z","iopub.execute_input":"2024-04-20T13:25:23.333181Z","iopub.status.idle":"2024-04-20T13:25:23.338814Z","shell.execute_reply.started":"2024-04-20T13:25:23.333144Z","shell.execute_reply":"2024-04-20T13:25:23.337567Z"}}},{"cell_type":"code","source":"torch_script_fp32 = torch.jit.optimize_for_inference(torch.jit.script(model))","metadata":{"execution":{"iopub.status.busy":"2024-04-20T13:25:44.992244Z","iopub.execute_input":"2024-04-20T13:25:44.993061Z","iopub.status.idle":"2024-04-20T13:25:47.174908Z","shell.execute_reply.started":"2024-04-20T13:25:44.993001Z","shell.execute_reply":"2024-04-20T13:25:47.173377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with torch.no_grad():\n    torch_script_fp32_time_ms = timeit.timeit(lambda: torch_script_fp32(CFG.DUMMY_INPUT_TENSOR), number=CFG.LOOP) / CFG.LOOP * 1000\nresult_tensor = torch_script_fp32(CFG.DUMMY_INPUT_TENSOR)\nprint(f'time: {torch_script_fp32_time_ms:.2f} msec (n={CFG.LOOP})')\nprint(f'mae:  {torch.mean(torch.abs(reference_tensor-result_tensor))}')","metadata":{"execution":{"iopub.status.busy":"2024-04-20T13:28:40.595998Z","iopub.execute_input":"2024-04-20T13:28:40.596461Z","iopub.status.idle":"2024-04-20T13:28:45.057281Z","shell.execute_reply.started":"2024-04-20T13:28:40.596428Z","shell.execute_reply":"2024-04-20T13:28:45.056031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ONNX FP32","metadata":{}},{"cell_type":"code","source":"torch.onnx.export(model,\n                  CFG.DUMMY_INPUT_TENSOR,\n                  CFG.OUTPUT_DIR_ONNX / 'fp32.onnx',\n                  opset_version=15,\n                  input_names=['input'],\n                  output_names=['output'])","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:28:50.667295Z","iopub.execute_input":"2024-04-20T09:28:50.667728Z","iopub.status.idle":"2024-04-20T09:28:53.441938Z","shell.execute_reply.started":"2024-04-20T09:28:50.667694Z","shell.execute_reply":"2024-04-20T09:28:53.440435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session = rt.InferenceSession(CFG.OUTPUT_DIR_ONNX / 'fp32.onnx')\n[session.run([], {'input': CFG.DUMMY_INPUT_NUMPY_FP32}) for i in range(5)]\nonnx_fp32_time_ms = timeit.timeit(lambda: session.run([], {'input': CFG.DUMMY_INPUT_NUMPY_FP32}), number=CFG.LOOP) / CFG.LOOP * 1000\nresult_numpy = session.run([], {'input': CFG.DUMMY_INPUT_NUMPY_FP32})[0]\nprint(f'time: {onnx_fp32_time_ms:.2f} msec (n={CFG.LOOP})')\nprint(f'mae:  {np.mean(np.abs(reference_numpy-result_numpy))}')","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:28:53.444581Z","iopub.execute_input":"2024-04-20T09:28:53.444993Z","iopub.status.idle":"2024-04-20T09:28:54.189910Z","shell.execute_reply.started":"2024-04-20T09:28:53.444958Z","shell.execute_reply":"2024-04-20T09:28:54.188655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ONNX FP32 (onnxsim)\n\nhttps://github.com/daquexian/onnx-simplifier","metadata":{}},{"cell_type":"code","source":"!onnxsim {(CFG.OUTPUT_DIR_ONNX / 'fp32.onnx').as_posix()} {(CFG.OUTPUT_DIR_ONNX / 'fp32_sim.onnx').as_posix()}","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:28:54.192022Z","iopub.execute_input":"2024-04-20T09:28:54.192809Z","iopub.status.idle":"2024-04-20T09:28:57.672910Z","shell.execute_reply.started":"2024-04-20T09:28:54.192747Z","shell.execute_reply":"2024-04-20T09:28:57.671418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session = rt.InferenceSession((CFG.OUTPUT_DIR_ONNX / 'fp32_sim.onnx'))\n[session.run([], {'input': CFG.DUMMY_INPUT_NUMPY_FP32}) for i in range(5)]\nonnx_fp32_sim_time_ms = timeit.timeit(lambda: session.run([], {'input': CFG.DUMMY_INPUT_NUMPY_FP32}), number=CFG.LOOP) / CFG.LOOP * 1000\nresult_numpy = session.run([], {'input': CFG.DUMMY_INPUT_NUMPY_FP32})[0]\nprint(f'time: {onnx_fp32_sim_time_ms:.2f} msec (n={CFG.LOOP})')\nprint(f'mae:  {np.mean(np.abs(reference_numpy-result_numpy))}')","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:28:57.678656Z","iopub.execute_input":"2024-04-20T09:28:57.679074Z","iopub.status.idle":"2024-04-20T09:28:58.484672Z","shell.execute_reply.started":"2024-04-20T09:28:57.679037Z","shell.execute_reply":"2024-04-20T09:28:58.482439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ONNX FP16\n\nhttps://onnxruntime.ai/docs/performance/model-optimizations/float16.html","metadata":{}},{"cell_type":"code","source":"model_fp32 = onnx.load(CFG.OUTPUT_DIR_ONNX / 'fp32.onnx')\nmodel_fp16 = float16.convert_float_to_float16(model_fp32)\nonnx.save(model_fp16, CFG.OUTPUT_DIR_ONNX / 'fp16.onnx')","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:28:58.486833Z","iopub.execute_input":"2024-04-20T09:28:58.487332Z","iopub.status.idle":"2024-04-20T09:28:58.960767Z","shell.execute_reply.started":"2024-04-20T09:28:58.487289Z","shell.execute_reply":"2024-04-20T09:28:58.959409Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"session = rt.InferenceSession(CFG.OUTPUT_DIR_ONNX / 'fp16.onnx')\n[session.run([], {'input': CFG.DUMMY_INPUT_NUMPY_FP16}) for i in range(5)]\nonnx_fp16_time_ms = timeit.timeit(lambda: session.run([], {'input': CFG.DUMMY_INPUT_NUMPY_FP16}), number=CFG.LOOP) / CFG.LOOP * 1000\nresult_numpy = session.run([], {'input': CFG.DUMMY_INPUT_NUMPY_FP16})[0]\nprint(f'time: {onnx_fp16_time_ms:.2f} msec (n={CFG.LOOP})')\nprint(f'mae:  {np.mean(np.abs(reference_numpy-result_numpy))}')","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:28:58.962503Z","iopub.execute_input":"2024-04-20T09:28:58.962965Z","iopub.status.idle":"2024-04-20T09:29:01.269308Z","shell.execute_reply.started":"2024-04-20T09:28:58.962922Z","shell.execute_reply":"2024-04-20T09:29:01.268089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## OpenVINO FP32","metadata":{}},{"cell_type":"code","source":"def infernce_openvino(\n    infer_request: openvino.runtime.ie_api.InferRequest,\n    input_data: np.ndarray\n) -> openvino.runtime.utils.data_helpers.wrappers.OVDict:\n    input_ov_tensor = ov.Tensor(array=input_data, shared_memory=True)\n    infer_request.set_input_tensor(input_ov_tensor)\n    output = infer_request.infer()\n    # If asynchronous processing is used effectively, there is a potential to further speed up the inference workflow.\n    # infer_request.start_async()\n    # infer_request.wait()\n    # output_numpy = infer_request.get_output_tensor().data\n    return output","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:59:50.104018Z","iopub.execute_input":"2024-04-20T09:59:50.104637Z","iopub.status.idle":"2024-04-20T09:59:50.114046Z","shell.execute_reply.started":"2024-04-20T09:59:50.104578Z","shell.execute_reply":"2024-04-20T09:59:50.112587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ov_model = ov.convert_model(CFG.OUTPUT_DIR_ONNX / 'fp32.onnx',\n                            input=[('input', CFG.INPUT_SHAPE)],)\nov.save_model(ov_model, CFG.OUTPUT_DIR_OV / 'fp32.xml', compress_to_fp16=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:29:01.281357Z","iopub.execute_input":"2024-04-20T09:29:01.281902Z","iopub.status.idle":"2024-04-20T09:29:01.757077Z","shell.execute_reply.started":"2024-04-20T09:29:01.281864Z","shell.execute_reply":"2024-04-20T09:29:01.755898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"core = ov.Core()\ncompiled_model = core.compile_model(CFG.OUTPUT_DIR_OV / 'fp32.xml', device_name='CPU')\ninfer_request = compiled_model.create_infer_request()\n[infernce_openvino(infer_request, CFG.DUMMY_INPUT_NUMPY_FP32) for i in range(5)]\nopenvino_fp32_time_ms = timeit.timeit(lambda: infernce_openvino(infer_request, CFG.DUMMY_INPUT_NUMPY_FP32), number=CFG.LOOP) / CFG.LOOP * 1000\nresult_numpy = infernce_openvino(infer_request, CFG.DUMMY_INPUT_NUMPY_FP32)['output']\nprint(f'time: {openvino_fp32_time_ms:.2f} msec (n={CFG.LOOP})')\nprint(f'mae:  {np.mean(np.abs(reference_numpy-result_numpy))}')","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:29:01.759503Z","iopub.execute_input":"2024-04-20T09:29:01.759861Z","iopub.status.idle":"2024-04-20T09:29:02.527974Z","shell.execute_reply.started":"2024-04-20T09:29:01.759832Z","shell.execute_reply":"2024-04-20T09:29:02.526820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## OpenVINO FP16","metadata":{"execution":{"iopub.status.busy":"2024-04-19T18:47:46.202412Z","iopub.execute_input":"2024-04-19T18:47:46.202895Z","iopub.status.idle":"2024-04-19T18:47:46.209934Z","shell.execute_reply.started":"2024-04-19T18:47:46.202858Z","shell.execute_reply":"2024-04-19T18:47:46.208503Z"}}},{"cell_type":"code","source":"ov_model = ov.convert_model(CFG.OUTPUT_DIR_ONNX / 'fp32.onnx',\n                            input=[('input', CFG.INPUT_SHAPE)],)\nov.save_model(ov_model, CFG.OUTPUT_DIR_OV / 'fp16.xml', compress_to_fp16=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:29:02.529510Z","iopub.execute_input":"2024-04-20T09:29:02.529944Z","iopub.status.idle":"2024-04-20T09:29:02.963102Z","shell.execute_reply.started":"2024-04-20T09:29:02.529906Z","shell.execute_reply":"2024-04-20T09:29:02.961784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"core = ov.Core()\ncompiled_model = core.compile_model(CFG.OUTPUT_DIR_OV / 'fp16.xml', device_name='CPU')\ninfer_request = compiled_model.create_infer_request()\n# The precision of ndarray is fine with FP32.\n[infernce_openvino(infer_request, CFG.DUMMY_INPUT_NUMPY_FP32) for i in range(5)]\nopenvino_fp16_time_ms = timeit.timeit(lambda: infernce_openvino(infer_request, CFG.DUMMY_INPUT_NUMPY_FP32), number=CFG.LOOP) / CFG.LOOP * 1000\nresult_numpy = infernce_openvino(infer_request, CFG.DUMMY_INPUT_NUMPY_FP32)['output']\nprint(f'time: {openvino_fp16_time_ms:.2f} msec (n={CFG.LOOP})')\nprint(f'mae:  {np.mean(np.abs(reference_numpy-result_numpy))}')","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:29:02.965776Z","iopub.execute_input":"2024-04-20T09:29:02.966160Z","iopub.status.idle":"2024-04-20T09:29:03.744017Z","shell.execute_reply.started":"2024-04-20T09:29:02.966122Z","shell.execute_reply":"2024-04-20T09:29:03.742794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compare","metadata":{}},{"cell_type":"markdown","source":"## Batch size","metadata":{}},{"cell_type":"code","source":"for batch_size in [1, 2, 4, 8, 16, 32]:\n    input_shape = [batch_size] + CFG.INPUT_SHAPE[1:]\n\n    # ONNX FP32\n    onnx_fp32_path = CFG.OUTPUT_DIR_ONNX / f\"fp32-bs{batch_size}.onnx\"\n    torch.onnx.export(model, torch.randn(*input_shape), onnx_fp32_path, opset_version=15, input_names=['input'], output_names=['output']) \n\n    # ONNX FP32 (sim)\n    onnx_fp32_sim_path = CFG.OUTPUT_DIR_ONNX / f\"fp32-bs{batch_size}_sim.onnx\"\n    !onnxsim {onnx_fp32_path.as_posix()} {onnx_fp32_sim_path.as_posix()}\n\n    # OpenVINO FP32\n    openvino_fp32_path = CFG.OUTPUT_DIR_OV / f\"fp32-bs{batch_size}.xml\"\n    ov_model = ov.convert_model(onnx_fp32_path, input=[('input', input_shape)])\n    ov.save_model(ov_model, openvino_fp32_path, compress_to_fp16=False)\n\n    # OpenVINO FP32 (sim)\n    openvino_fp32_sim_path = CFG.OUTPUT_DIR_OV / f\"fp32-bs{batch_size}_sim.xml\"\n    ov_model = ov.convert_model(onnx_fp32_sim_path, input=[('input', input_shape)])\n    ov.save_model(ov_model, openvino_fp32_sim_path, compress_to_fp16=False)\n\n    # OpenVINO FP16\n    openvino_fp16_path = CFG.OUTPUT_DIR_OV / f\"fp16-bs{batch_size}.xml\"\n    ov_model = ov.convert_model(onnx_fp32_path, input=[('input', input_shape)])\n    ov.save_model(ov_model, openvino_fp16_path, compress_to_fp16=True)\n\n    # OpenVINO FP16 (sim)\n    openvino_fp16_sim_path = CFG.OUTPUT_DIR_OV / f\"fp16-bs{batch_size}_sim.xml\"\n    ov_model = ov.convert_model(onnx_fp32_sim_path, input=[('input', input_shape)])\n    ov.save_model(ov_model, openvino_fp16_sim_path, compress_to_fp16=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T09:29:03.745694Z","iopub.execute_input":"2024-04-20T09:29:03.746065Z","iopub.status.idle":"2024-04-20T09:29:53.223788Z","shell.execute_reply.started":"2024-04-20T09:29:03.746034Z","shell.execute_reply":"2024-04-20T09:29:53.222449Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = []\nfor batch_size in [1, 2, 4, 8, 16, 32]:\n    onnx_fp32_path = CFG.OUTPUT_DIR_ONNX / f\"fp32-bs{batch_size}.onnx\"\n    onnx_fp32_sim_path = CFG.OUTPUT_DIR_ONNX / f\"fp32-bs{batch_size}_sim.onnx\"\n    openvino_fp32_path = CFG.OUTPUT_DIR_OV / f\"fp32-bs{batch_size}.xml\"\n    openvino_fp32_sim_path = CFG.OUTPUT_DIR_OV / f\"fp32-bs{batch_size}_sim.xml\"\n    openvino_fp16_path = CFG.OUTPUT_DIR_OV / f\"fp16-bs{batch_size}.xml\"\n    openvino_fp16_sim_path = CFG.OUTPUT_DIR_OV / f\"fp16-bs{batch_size}_sim.xml\"\n\n    for i in range(CFG.REPEAT):\n        input_shape = [batch_size] + CFG.INPUT_SHAPE[1:]\n        dummy_input_tensor_fp32 = torch.randn(CFG.LOOP, *input_shape)\n        dummy_input_numpy_fp32 = dummy_input_tensor_fp32.numpy()\n\n        # PyTorch FP32\n        with torch.no_grad():\n            [model(dummy_input_tensor_fp32[0]) for i in range(3)]\n            for j in range(CFG.LOOP):\n                count = i * CFG.LOOP + j\n                start = time.perf_counter()\n                result = model(dummy_input_tensor_fp32[j])\n                end = time.perf_counter()\n                time_ms = (end - start) * 1000\n                data.append(['pytorch_fp32', batch_size, count, time_ms, time_ms / batch_size])\n\n        # Torch Script FP32\n        with torch.no_grad():\n            [torch_script_fp32(dummy_input_tensor_fp32[0]) for i in range(3)]\n            for j in range(CFG.LOOP):\n                count = i * CFG.LOOP + j\n                start = time.perf_counter()\n                result = torch_script_fp32(dummy_input_tensor_fp32[j])\n                end = time.perf_counter()\n                time_ms = (end - start) * 1000\n                data.append(['torch_script_fp32', batch_size, count, time_ms, time_ms / batch_size])            \n\n        # ONNX FP32\n        session = rt.InferenceSession(onnx_fp32_path)\n        [session.run([], {'input': dummy_input_numpy_fp32[0]}) for i in range(3)]\n        for j in range(CFG.LOOP):\n            count = i * CFG.LOOP + j\n            start = time.perf_counter()\n            result = session.run([], {'input': dummy_input_numpy_fp32[j]})\n            end = time.perf_counter()\n            time_ms = (end - start) * 1000\n            data.append(['onnx_fp32', batch_size, count, time_ms, time_ms / batch_size])\n        session = None\n\n        # ONNX FP32 (sim)\n        session = rt.InferenceSession(onnx_fp32_sim_path)\n        [session.run([], {'input': dummy_input_numpy_fp32[0]}) for i in range(3)]\n        for j in range(CFG.LOOP):\n            count = i * CFG.LOOP + j\n            start = time.perf_counter()\n            result = session.run([], {'input': dummy_input_numpy_fp32[j]})\n            end = time.perf_counter()\n            time_ms = (end - start) * 1000\n            data.append(['onnx_fp32_sim', batch_size, count, time_ms, time_ms / batch_size])\n        session = None\n\n        # OpenVINO FP32\n        core = ov.Core()\n        compiled_model = core.compile_model(openvino_fp32_path, device_name='CPU')\n        infer_request = compiled_model.create_infer_request()\n        [infernce_openvino(infer_request, dummy_input_numpy_fp32[0]) for i in range(3)]\n        for j in range(CFG.LOOP):\n            count = i * CFG.LOOP + j\n            start = time.perf_counter()\n            result = infernce_openvino(infer_request, dummy_input_numpy_fp32[j])\n            end = time.perf_counter()\n            time_ms = (end - start) * 1000\n            data.append(['openvino_fp32', batch_size, count, time_ms, time_ms / batch_size])\n        core = compiled_model = infer_request = None\n\n        # OpenVINO FP32 (sim)\n        core = ov.Core()\n        compiled_model = core.compile_model(openvino_fp32_sim_path, device_name='CPU')\n        infer_request = compiled_model.create_infer_request()\n        [infernce_openvino(infer_request, dummy_input_numpy_fp32[0]) for i in range(3)]\n        for j in range(CFG.LOOP):\n            count = i * CFG.LOOP + j\n            start = time.perf_counter()\n            result = infernce_openvino(infer_request, dummy_input_numpy_fp32[j])\n            end = time.perf_counter()\n            time_ms = (end - start) * 1000\n            data.append(['openvino_fp32_sim', batch_size, count, time_ms, time_ms / batch_size])\n        core = compiled_model = infer_request = None\n\n        # OpenVINO FP16\n        core = ov.Core()\n        compiled_model = core.compile_model(openvino_fp16_path, device_name='CPU')\n        infer_request = compiled_model.create_infer_request()\n        [infernce_openvino(infer_request, dummy_input_numpy_fp32[0]) for i in range(3)]\n        for j in range(CFG.LOOP):\n            count = i * CFG.LOOP + j\n            start = time.perf_counter()\n            result = infernce_openvino(infer_request, dummy_input_numpy_fp32[j])\n            end = time.perf_counter()\n            time_ms = (end - start) * 1000\n            data.append(['openvino_fp16', batch_size, count, time_ms, time_ms / batch_size])\n        core = compiled_model = infer_request = None\n\n        # OpenVINO FP16 (sim)\n        core = ov.Core()\n        compiled_model = core.compile_model(openvino_fp16_sim_path, device_name='CPU')\n        infer_request = compiled_model.create_infer_request()\n        [infernce_openvino(infer_request, dummy_input_numpy_fp32[0]) for i in range(3)]\n        for j in range(CFG.LOOP):\n            count = i * CFG.LOOP + j\n            start = time.perf_counter()\n            result = infernce_openvino(infer_request, dummy_input_numpy_fp32[j])\n            end = time.perf_counter()\n            time_ms = (end - start) * 1000\n            data.append(['openvino_fp16_sim', batch_size, count, time_ms, time_ms / batch_size])\n        core = compiled_model = infer_request = None\n\ndf = pd.DataFrame(data=data, columns=['type', 'batch_size', 'count', 'time_ms', 'time_per_image_ms'])\ndf.to_csv('time.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T13:34:08.152739Z","iopub.execute_input":"2024-04-20T13:34:08.153210Z","iopub.status.idle":"2024-04-20T13:35:02.019763Z","shell.execute_reply.started":"2024-04-20T13:34:08.153177Z","shell.execute_reply":"2024-04-20T13:35:02.018351Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(20, 5))\nax = sns.barplot(data=df, x='batch_size', y='time_ms', hue='type')\nfor container in ax.containers:\n    ax.bar_label(container, fmt='%.1f', fontsize=7)\nax.set_title(f'Inference time for n={CFG.REPEAT * CFG.LOOP} (milliseconds)')\nax.legend(loc=\"upper left\", bbox_to_anchor=(1.02, 1.0), borderaxespad=0)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T13:36:21.899093Z","iopub.execute_input":"2024-04-20T13:36:21.899569Z","iopub.status.idle":"2024-04-20T13:36:22.996709Z","shell.execute_reply.started":"2024-04-20T13:36:21.899529Z","shell.execute_reply":"2024-04-20T13:36:22.995442Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **sim:** Model optimized using onnxsim.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(20, 5))\nax = sns.barplot(data=df, x='batch_size', y='time_per_image_ms', hue='type')\nfor container in ax.containers:\n    ax.bar_label(container, fmt='%.1f', fontsize=7)\nax.set_title(f'Inference time per image for n={CFG.REPEAT * CFG.LOOP} (milliseconds)')\nax.legend(loc=\"upper left\", bbox_to_anchor=(1.02, 1.0), borderaxespad=0)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T13:36:42.163030Z","iopub.execute_input":"2024-04-20T13:36:42.163529Z","iopub.status.idle":"2024-04-20T13:36:43.225560Z","shell.execute_reply.started":"2024-04-20T13:36:42.163471Z","shell.execute_reply":"2024-04-20T13:36:43.224380Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **sim:** Model optimized using onnxsim.","metadata":{"execution":{"iopub.status.busy":"2024-04-20T11:27:28.624520Z","iopub.execute_input":"2024-04-20T11:27:28.624962Z","iopub.status.idle":"2024-04-20T11:27:28.632954Z","shell.execute_reply.started":"2024-04-20T11:27:28.624927Z","shell.execute_reply":"2024-04-20T11:27:28.631099Z"}}},{"cell_type":"markdown","source":"## Sync vs. Async\n\nhttps://docs.openvino.ai/2023.3/openvino_docs_OV_UG_Python_API_exclusives.html","metadata":{}},{"cell_type":"code","source":"dummy_input_numpy_fp32 = torch.randn(CFG.LOOP, *CFG.INPUT_SHAPE).numpy()\n\n# OpenVINO FP16 BS-4 Async\nresult_async = []\ncore = ov.Core()\ncompiled_model = core.compile_model(CFG.OUTPUT_DIR_OV / \"fp16-bs4.xml\", device_name='CPU')\njobs = 4\ninfer_queue = ov.AsyncInferQueue(compiled_model, jobs)\n\nchunks = [dummy_input_numpy_fp32[i:i + jobs] for i in range(0, dummy_input_numpy_fp32.shape[0], jobs)]\nfor chunk in chunks:\n    for i in range(chunk.shape[0]):\n        infer_queue.start_async({'input': chunk[i]})\n    infer_queue.wait_all()\n    result_async.extend([infer_queue[i].get_output_tensor().data.copy() for i in range(jobs)])\n\n# OpenVINO FP16 BS-4 Sync\nresult_sync = []\ncore = ov.Core()\ncompiled_model = core.compile_model(CFG.OUTPUT_DIR_OV / \"fp16-bs4.xml\", device_name='CPU')\ninfer_request = compiled_model.create_infer_request()\nfor i in range(CFG.LOOP):\n    result_sync.append(infernce_openvino(infer_request, dummy_input_numpy_fp32[i])['output'])\ncore = compiled_model = infer_request = None\n\nprint([(result_async[i] == result_sync[i]).all() for i in range(CFG.LOOP)])","metadata":{"execution":{"iopub.status.busy":"2024-04-20T11:22:39.500745Z","iopub.execute_input":"2024-04-20T11:22:39.501156Z","iopub.status.idle":"2024-04-20T11:22:40.306164Z","shell.execute_reply.started":"2024-04-20T11:22:39.501125Z","shell.execute_reply":"2024-04-20T11:22:40.304666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = []\nfor batch_size in [2, 4, 8, 16, 32]:\n    openvino_fp16_path = CFG.OUTPUT_DIR_OV / f\"fp16-bs{batch_size}.xml\"\n    input_shape = [batch_size] + CFG.INPUT_SHAPE[1:]\n    for jobs in [0, 2, 4, 8]:\n        for i in range(CFG.REPEAT):\n            dummy_input_numpy_fp32 = torch.randn(CFG.LOOP, *input_shape).numpy()\n\n            if jobs == 0:\n                # OpenVINO FP16 Sync\n                core = ov.Core()\n                compiled_model = core.compile_model(openvino_fp16_path, device_name='CPU')\n                infer_request = compiled_model.create_infer_request()\n                [infernce_openvino(infer_request, dummy_input_numpy_fp32[0]) for i in range(3)]\n\n                for j in range(CFG.LOOP):\n                    count = i * CFG.LOOP + j\n                    start = time.perf_counter()\n                    result = infernce_openvino(infer_request, dummy_input_numpy_fp32[j])\n                    end = time.perf_counter()\n                    time_ms = (end - start) * 1000\n                    data.append([f'openvino_fp16_bs-{batch_size}', jobs, count, time_ms, time_ms / batch_size])\n                core = compiled_model = infer_request = None\n            else:\n                # OpenVINO FP16 Async\n                core = ov.Core()\n                compiled_model = core.compile_model(openvino_fp16_path, device_name='CPU')\n                infer_queue = ov.AsyncInferQueue(compiled_model, jobs)\n\n                chunks = [dummy_input_numpy_fp32[i:i + jobs] for i in range(0, dummy_input_numpy_fp32.shape[0], jobs)]\n                for j, chunk in enumerate(chunks):\n                    count = i * CFG.LOOP + j * jobs\n                    start = time.perf_counter()\n                    for k in range(chunk.shape[0]):\n                        infer_queue.start_async({'input': chunk[k]})\n                    infer_queue.wait_all()\n                    results = [infer_queue[i].get_output_tensor().data.copy() for i in range(jobs)]\n                    end = time.perf_counter()\n                    time_ms = (end - start) * 1000\n                    data.append([f'openvino_fp16_bs-{batch_size}', jobs, count, time_ms / len(chunk), time_ms / len(chunk) / batch_size])\n                core = compiled_model = infer_queue = None\n\ndf = pd.DataFrame(data=data, columns=['type', 'jobs', 'count', 'time_ms', 'time_per_image_ms'])\ndf.to_csv('sync_vs_async.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T11:23:48.610371Z","iopub.execute_input":"2024-04-20T11:23:48.611648Z","iopub.status.idle":"2024-04-20T11:24:05.870710Z","shell.execute_reply.started":"2024-04-20T11:23:48.611606Z","shell.execute_reply":"2024-04-20T11:24:05.869160Z"},"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16, 5))\nax = sns.barplot(data=df, x='type', y='time_per_image_ms', hue='jobs')\nfor container in ax.containers:\n    ax.bar_label(container, fmt='%.1f', fontsize=7)\nax.set_title(f'Inference time per image using OpenVINO FP16 (milliseconds)')\nax.legend(title='jobs', loc=\"upper left\", bbox_to_anchor=(1.02, 1.0), borderaxespad=0)","metadata":{"execution":{"iopub.status.busy":"2024-04-20T11:26:12.380718Z","iopub.execute_input":"2024-04-20T11:26:12.381258Z","iopub.status.idle":"2024-04-20T11:26:13.124191Z","shell.execute_reply.started":"2024-04-20T11:26:12.381221Z","shell.execute_reply":"2024-04-20T11:26:13.122831Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **jobs==0:** Sync\n* **jobs>0:** Async\n* 🤔","metadata":{}},{"cell_type":"markdown","source":"# References\n\n* https://www.kaggle.com/competitions/birdclef-2024/discussion/492649\n* https://www.kaggle.com/code/honglihang/openvino-is-all-you-need\n* https://docs.openvino.ai/2024/notebooks/121-convert-to-openvino-with-output.html\n* https://docs.openvino.ai/2023.3/openvino_docs_model_processing_introduction.html\n* https://docs.openvino.ai/2023.3/openvino_docs_OV_UG_Python_API_exclusives.html\n* https://onnxruntime.ai/docs/performance/model-optimizations/float16.html\n* https://github.com/daquexian/onnx-simplifier","metadata":{}}]}