{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install onnx-tf\n!pip install onnxsim\n# !pip install tflite-runtime==2.9.1\n!pip install tflite-runtime","metadata":{"papermill":{"duration":31.150628,"end_time":"2023-04-13T01:32:28.450686","exception":false,"start_time":"2023-04-13T01:31:57.300058","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-13T05:39:30.918063Z","iopub.execute_input":"2023-04-13T05:39:30.918645Z","iopub.status.idle":"2023-04-13T05:40:00.452923Z","shell.execute_reply.started":"2023-04-13T05:39:30.918609Z","shell.execute_reply":"2023-04-13T05:40:00.451627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings(action='ignore')\n\nimport os\nimport gc\n\nimport json\nfrom tqdm import tqdm\nimport numpy as np\nimport pandas as pd\n\nimport torch\nprint(\"pytorch version:\",torch.__version__)\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nimport onnx\nimport onnxsim\nfrom onnx_tf.backend import prepare\n\nimport tflite_runtime\nprint(\"tflite_runtime:\",tflite_runtime.__version__)","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":26.677701,"end_time":"2023-04-13T01:32:55.133269","exception":false,"start_time":"2023-04-13T01:32:28.455568","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-13T05:40:00.456488Z","iopub.execute_input":"2023-04-13T05:40:00.457251Z","iopub.status.idle":"2023-04-13T05:40:00.467149Z","shell.execute_reply.started":"2023-04-13T05:40:00.457200Z","shell.execute_reply":"2023-04-13T05:40:00.465770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tensorflow Conversion","metadata":{"papermill":{"duration":0.005167,"end_time":"2023-04-13T01:32:55.143962","exception":false,"start_time":"2023-04-13T01:32:55.138795","status":"completed"},"tags":[]}},{"cell_type":"code","source":"class InputNet(nn.Module):\n    def __init__(self, max_length=220):\n        super().__init__()\n        self.max_length = max_length\n        self.LIP = [\n            61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n            291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n            78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n            95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n        ]\n        self.LHAND = list(range(468,489))\n        self.RHAND = list(range(522,543))\n    \n    def pre_process(self, xyz):\n        xyz = xyz - xyz[~torch.isnan(xyz)].mean(0,keepdims=True) #noramlisation to common maen\n        xyz = xyz / xyz[~torch.isnan(xyz)].std(0, keepdims=True)\n        lip   = xyz[:, self.LIP]\n        lhand = xyz[:, self.LHAND]\n        rhand = xyz[:, self.RHAND]\n        xyz = torch.cat([ #(none, 82, 3)\n            lip,\n            lhand,\n            rhand,\n        ], axis = 1)\n        xyz[torch.isnan(xyz)] = torch.tensor(0.0, dtype=torch.float32)\n        return xyz[:self.max_length]\n    \n    def forward(self, xyz):\n        if isinstance(xyz, np.ndarray):\n            xyz = torch.tensor(xyz)\n        elif isinstance(xyz,torch.Tensor):\n            pass\n        else:\n            raise TypeError\n        return self.pre_process(xyz)","metadata":{"papermill":{"duration":0.023241,"end_time":"2023-04-13T01:32:55.172646","exception":false,"start_time":"2023-04-13T01:32:55.149405","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-13T05:40:00.468949Z","iopub.execute_input":"2023-04-13T05:40:00.469332Z","iopub.status.idle":"2023-04-13T05:40:00.484833Z","shell.execute_reply.started":"2023-04-13T05:40:00.469296Z","shell.execute_reply":"2023-04-13T05:40:00.483775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def positional_encoding(length, embed_dim):\n    dim = embed_dim//2\n    position = np.arange(length)[:, np.newaxis]     # (seq, 1)\n    dim = np.arange(dim)[np.newaxis, :]/dim   # (1, dim)\n    angle = 1 / (10000**dim)         # (1, dim)\n    angle = position * angle    # (pos, dim)\n    pos_embed = np.concatenate(\n        [np.sin(angle), np.cos(angle)],\n        axis=-1\n    )\n    pos_embed = torch.from_numpy(pos_embed).float()\n    return pos_embed\n\n\nclass FeedForward(nn.Module):\n    def __init__(self, embed_dim, hidden_dim):\n        super().__init__()\n        self.mlp = nn.Sequential(\n            nn.Linear(embed_dim, hidden_dim),\n            nn.ReLU(inplace=True),\n            nn.Linear(hidden_dim, embed_dim),\n        )\n    def forward(self, x):\n        return self.mlp(x)\n\n\n#https://pytorch.org/docs/stable/generated/torch.nn.MultiheadAttention.html\nclass MultiHeadAttention(nn.Module):\n    def __init__(self,\n            embed_dim,\n            num_head,\n            batch_first,\n        ):\n        super().__init__()\n        self.embed_dim = embed_dim\n        self.num_head = num_head\n        self.mha = nn.MultiheadAttention(\n            embed_dim,\n            num_heads=num_head,\n            bias=True,\n            add_bias_kv=False,\n            kdim=None,\n            vdim=None,\n            dropout=0.0,\n            batch_first=batch_first,\n        )\n        \n\n    def forward(self, x):\n        q = F.linear(x[:1], self.mha.in_proj_weight[:self.embed_dim*1], self.mha.in_proj_bias[:self.embed_dim*1]) #since we need only cls\n        k = F.linear(x, self.mha.in_proj_weight[self.embed_dim*1:self.embed_dim*2], self.mha.in_proj_bias[self.embed_dim*1:self.embed_dim*2])\n        v = F.linear(x, self.mha.in_proj_weight[self.embed_dim*2:], self.mha.in_proj_bias[self.embed_dim*2:]) \n        q = q.reshape(-1, self.num_head, self.embed_dim//self.num_head).permute(1, 0, 2)\n        k = k.reshape(-1, self.num_head, self.embed_dim//self.num_head).permute(1, 2, 0)\n        v = v.reshape(-1, self.num_head, self.embed_dim//self.num_head).permute(1, 0, 2)\n        dot  = torch.matmul(q, k) * (1/(self.embed_dim//self.num_head)**0.5) # H L L\n        attn = F.softmax(dot, -1)  #   L L\n        out  = torch.matmul(attn, v)  #   L H dim\n        out  = out.permute(1, 0, 2).reshape(-1, self.embed_dim)\n        out  = F.linear(out, self.mha.out_proj.weight, self.mha.out_proj.bias)  \n        return out\n\nclass TransformerBlock(nn.Module):\n    def __init__(self,\n        embed_dim,\n        num_head,\n        out_dim,\n        batch_first=True,\n    ):\n        super().__init__()\n        self.attn  = MultiHeadAttention(embed_dim, num_head,batch_first)\n        self.ffn   = FeedForward(embed_dim, out_dim)\n        self.norm1 = nn.LayerNorm(embed_dim)\n        self.norm2 = nn.LayerNorm(out_dim)\n\n    def forward(self, x):\n        x = x[:1] + self.attn((self.norm1(x))) #확인\n        x = x + self.ffn((self.norm2(x)))\n        return x\n\n\nclass Net(nn.Module):\n    def __init__(\n            self,\n            num_class=250,\n            embed_dim=512,\n            max_length=220,\n            num_head=4,\n            num_layer=1,\n            p=0.4,\n            s_pose=False,\n            add_dxyz=False,\n            mean_pool=False,\n            **kwargs\n        ):\n        super().__init__()\n        self.num_joint = 115 if s_pose else 82\n        self.max_length = max_length\n        self.p = p\n        self.joint_dim = 3*(1 + add_dxyz + mean_pool)\n\n        pos_embed = positional_encoding(self.max_length, embed_dim)\n        # self.register_buffer('pos_embed', pos_embed)\n        self.pos_embed = nn.Parameter(pos_embed)\n\n        self.cls_embed = nn.Parameter(torch.zeros((1, embed_dim)))\n        self.x_embed = nn.Sequential(\n            nn.Linear(self.num_joint * self.joint_dim, embed_dim, bias=False),\n        )\n\n        self.encoder = nn.ModuleList([\n            TransformerBlock(\n                embed_dim,\n                num_head,\n                embed_dim,\n            ) for _ in range(num_layer)\n        ])\n        self.logit = nn.Linear(embed_dim, num_class)\n        self.get_device()\n\n    def get_device(self):\n        self.device = next(self.parameters()).device\n    \n    def forward(self, xyz):\n        L = xyz.shape[0]\n        x_embed = self.x_embed(xyz.flatten(1))\n        x = x_embed[:L] + self.pos_embed[:L]\n        x = torch.cat([self.cls_embed, x],0)\n        x = self.encoder[0](x)\n        cls = x[[0]]\n        logit = self.logit(cls)\n        return logit","metadata":{"papermill":{"duration":0.035868,"end_time":"2023-04-13T01:32:55.213987","exception":false,"start_time":"2023-04-13T01:32:55.178119","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-13T05:40:00.489882Z","iopub.execute_input":"2023-04-13T05:40:00.490197Z","iopub.status.idle":"2023-04-13T05:40:00.517165Z","shell.execute_reply.started":"2023-04-13T05:40:00.490171Z","shell.execute_reply":"2023-04-13T05:40:00.516148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROWS_PER_FRAME = 543\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"papermill":{"duration":0.016,"end_time":"2023-04-13T01:32:55.235384","exception":false,"start_time":"2023-04-13T01:32:55.219384","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-13T05:40:00.518559Z","iopub.execute_input":"2023-04-13T05:40:00.519743Z","iopub.status.idle":"2023-04-13T05:40:00.530374Z","shell.execute_reply.started":"2023-04-13T05:40:00.519701Z","shell.execute_reply":"2023-04-13T05:40:00.529264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 1:\n    #load model\n    device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n    print(\"device:\",device)\n\n    model_checkpoint_path = \"/kaggle/input/private-dataset2/2585.6619_best_model_TL1.5569_TA0.8668_VL1.9869_VA0.7351.pt\"\n    model_config_path = \"/kaggle/input/private-dataset2/config.json\"\n    with open(model_config_path, \"r\") as f:\n        model_config = json.load(f)\n    model_config['embed_dim'] = 16\n    model_config['max_length'] = 16\n    model = Net(**model_config)\n    model.to(device)\n    model.get_device()\n#     model_weights = torch.load(model_checkpoint_path).state_dict()\n#     model.load_state_dict(model_weights,strict=False)\n\n\n    #feature gen : pytorch to onnx\n    sample_input = torch.rand((1, 543, 3))\n    onnx_feat_gen_path = 'feature_gen.onnx'\n    input_net = InputNet(max_length=model_config['max_length'])\n    input_net.eval()\n\n    torch.onnx.export(\n        input_net,                  # PyTorch Model\n        sample_input,                    # Input tensor\n        onnx_feat_gen_path,        # Output file (eg. 'output_model.onnx')\n        export_params = True,     # store the trained parameter weights inside the model file\n        opset_version=12,       # Operator support version\n        do_constant_folding = True, # whether to execute constant folding for optimization \n        input_names=['input'],   # Input tensor name (arbitary)\n        output_names=['output'], # Output tensor name (arbitary)\n        dynamic_axes={\n            'input' : {0: 'num_frames'}\n        }\n    )\n\n    #model : pytorch to onnx\n    sample_input = torch.rand((1, 82, 3)).to(device)\n    onnx_model_path = 'asl_model.onnx'\n\n    model.eval()\n\n    torch.onnx.export(\n        model,                  # PyTorch Model\n        sample_input,                    # Input tensor\n        onnx_model_path,        # Output file (eg. 'output_model.onnx')\n        export_params = True, # store the trained parameter weights inside the model file\n        opset_version=12,       # Operator support version\n        do_constant_folding = True, # whether to execute constant folding for optimization \n        input_names=['input'],   # Input tensor name (arbitary)\n        output_names=['output'], # Output tensor name (arbitary)\n        dynamic_axes={\n            'input' : {\n                0: 'batch_size',\n                1: \"num_frames\",\n            }\n        }\n    )\n\n    for f in [onnx_feat_gen_path, onnx_model_path]:\n        if f is None: raise NameError\n        model = onnx.load(f)\n        onnx.checker.check_model(model)\n        model_simple, check = onnxsim.simplify(model)\n        onnx.save(model_simple, f)\n\n\n    #onnx to tflite\n    tf_feat_gen_path = '/kaggle/working/feature_gen'\n    onnx_feat_gen = onnx.load(onnx_feat_gen_path)\n    tf_rep = prepare(onnx_feat_gen)\n    tf_rep.export_graph(tf_feat_gen_path)\n\n\n    tf_model_path = '/kaggle/working/asl_model'\n    onnx_model = onnx.load(onnx_model_path)\n    tf_rep = prepare(onnx_model)\n    tf_rep.export_graph(tf_model_path)\n\n\n\n    import tensorflow as tf\n    tf_file = \"/kaggle/working/tf_infer_model\"\n    class ASLInferModel(tf.Module):\n        def __init__(self):\n            super(ASLInferModel, self).__init__()\n            self.feature_gen = tf.saved_model.load(tf_feat_gen_path)\n            self.model = tf.saved_model.load(tf_model_path)\n            self.feature_gen.trainable = False\n            self.model.trainable = False\n\n        @tf.function(input_signature=[\n          tf.TensorSpec(shape=[None, 543, 3], dtype=tf.float32, name='inputs')\n        ])\n        def call(self, input):\n            output_tensors = {}\n            features = self.feature_gen(**{'input': input})['output']\n            output_tensors['outputs'] = self.model(**{'input': features})['output'][0]\n            return output_tensors\n\n\n    mytfmodel = ASLInferModel()\n    tf.saved_model.save(mytfmodel, tf_file, signatures={'serving_default': mytfmodel.call})\n\n    # Convert the model\n    converter = tf.lite.TFLiteConverter.from_saved_model(tf_file)\n    # converter.target_spec.supported_ops = [\n    #   tf.lite.OpsSet.TFLITE_BUILTINS, # enable TensorFlow Lite ops.\n    #   tf.lite.OpsSet.SELECT_TF_OPS # enable TensorFlow ops.\n    # ]\n    tflite_model = converter.convert()\n\n    tflite_model_path = 'model.tflite'\n\n    # # Save the model\n    with open(tflite_model_path, 'wb') as f:\n        f.write(tflite_model)\n    print(\"tflite convert() passed\")","metadata":{"papermill":{"duration":0.031022,"end_time":"2023-04-13T01:32:55.272092","exception":false,"start_time":"2023-04-13T01:32:55.241070","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-13T05:40:00.532076Z","iopub.execute_input":"2023-04-13T05:40:00.532591Z","iopub.status.idle":"2023-04-13T05:40:08.398055Z","shell.execute_reply.started":"2023-04-13T05:40:00.532551Z","shell.execute_reply":"2023-04-13T05:40:08.395428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final Inference Model in Tensorflow","metadata":{"papermill":{"duration":0.005101,"end_time":"2023-04-13T01:32:55.282824","exception":false,"start_time":"2023-04-13T01:32:55.277723","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import tflite_runtime.interpreter as tflite\npq_path = \"/kaggle/input/asl-signs/train_landmark_files/53618/1001379621.parquet\"\ntflite_model_path = \"/kaggle/working/model.tflite\"\ninterpreter = tflite.Interpreter(tflite_model_path)\ninterpreter.allocate_tensors()\n\nfound_signatures = list(interpreter.get_signature_list().keys())\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")\n\noutput = prediction_fn(inputs=load_relevant_data_subset(pq_path))\nsign = np.argmax(output[\"outputs\"])\n\nprint(sign, output[\"outputs\"].shape)","metadata":{"papermill":{"duration":0.283604,"end_time":"2023-04-13T01:32:55.571784","exception":false,"start_time":"2023-04-13T01:32:55.288180","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-13T05:40:08.399624Z","iopub.execute_input":"2023-04-13T05:40:08.400622Z","iopub.status.idle":"2023-04-13T05:40:08.428182Z","shell.execute_reply.started":"2023-04-13T05:40:08.400580Z","shell.execute_reply":"2023-04-13T05:40:08.427133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mode = 'submit' #debug #submit\n\nimport pandas as pd\nimport numpy as np\nimport os\nimport shutil\nfrom datetime import datetime\nfrom timeit import default_timer as timer\n\n\nif mode in ['debug']:  \n    try:\n        import tflite_runtime\n    except:\n        !pip install tflite-runtime\n\n    import tflite_runtime.interpreter as tflite   \n    import tflite_runtime\n    print(tflite_runtime.__version__)\n    #'2.11.0'\n    \n    #import tensorflow as tf\n    #print(tf.__version__)\n    # 2.11.0\n\nprint('import ok')\n'''\nYour model must also require less than 40 MB in memory and \nperform inference with less than 100 milliseconds of latency per video. \nExpect to see approximately 40,000 videos in the test set. \nWe allow an additional 10 minute buffer for loading the data and miscellaneous overhead.\n\n'''\ndef time_to_str(t, mode='min'):\n    if mode=='min':\n        t  = int(t)/60\n        hr = t//60\n        min = t%60\n        return '%2d hr %02d min'%(hr,min)\n\n    elif mode=='sec':\n        t   = int(t)\n        min = t//60\n        sec = t%60\n        return '%2d min %02d sec'%(min,sec)\n\n    else:\n        raise NotImplementedError\n\n        \nROWS_PER_FRAME = 543\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)\n\nif mode in ['debug']: \n \n    interpreter = tflite.Interpreter(tflite_model_path)\n    prediction_fn = interpreter.get_signature_runner('serving_default')\n\n    valid_df = pd.read_csv('/kaggle/input/asl-demo/train_prepared.csv') \n    valid_df = valid_df[valid_df.fold==2].reset_index(drop=True)\n    valid_df = valid_df[:4_000]\n    valid_num = len(valid_df)\n    valid = {\n        'sign':[],\n    }\n\n    start_timer = timer()\n    for t, d in valid_df.iterrows():\n#         gc.collect()\n        pq_file = f'/kaggle/input/asl-signs/{d.path}'\n        #print(pq_file)\n        xyz = load_relevant_data_subset(pq_file)\n\n        output = prediction_fn(inputs=xyz)\n        p = output['outputs'].reshape(-1)\n\n        valid['sign'].append(p)\n\n        #---\n        if t%100==0:\n            time_taken = timer() - start_timer\n            print('\\r %8d / %d  %s'%(t,valid_num,time_to_str(time_taken,'sec')),end='',flush=True)\n\n    print('\\n')\n\n\n    truth = valid_df.label.values\n    sign  = np.stack(valid['sign'])\n    predict = np.argsort(-sign, -1)\n    correct = predict==truth.reshape(valid_num,1)\n    topk = correct.cumsum(-1).mean(0)[:5]\n\n\n    print(f'time_taken = {time_to_str(time_taken,\"sec\")}')\n    print(f'time_taken for LB = {time_taken*1000/valid_num:05f} msec\\n')\n    for i in range(5):\n        print(f'topk[{i}] = {topk[i]}')  \n    print('----- end -----\\n')","metadata":{"papermill":{"duration":137.448971,"end_time":"2023-04-13T01:35:13.026385","exception":false,"start_time":"2023-04-13T01:32:55.577414","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-13T05:40:08.429892Z","iopub.execute_input":"2023-04-13T05:40:08.430645Z","iopub.status.idle":"2023-04-13T05:40:08.469553Z","shell.execute_reply.started":"2023-04-13T05:40:08.430605Z","shell.execute_reply":"2023-04-13T05:40:08.468512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shutil.copyfile(tflite_model_path, 'model.tflite') \n# !zip submission.zip $tflite_model_path\n!zip submission.zip  'model.tflite'\n!ls\n\nprint('tflite_file:', tflite_model_path)\nprint(f'submit ok')","metadata":{"papermill":{"duration":0.993454,"end_time":"2023-04-13T01:35:14.028434","exception":false,"start_time":"2023-04-13T01:35:13.034980","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-13T05:40:08.471081Z","iopub.execute_input":"2023-04-13T05:40:08.471931Z","iopub.status.idle":"2023-04-13T05:40:10.544845Z","shell.execute_reply.started":"2023-04-13T05:40:08.471891Z","shell.execute_reply":"2023-04-13T05:40:10.543217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.009634,"end_time":"2023-04-13T01:35:14.046844","exception":false,"start_time":"2023-04-13T01:35:14.037210","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}