{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# End to End Pytorch:\n\nThis notebook trains a pytorch model or rather 2 pytorch models to do our work. The pre-processing part of the work which converts our data to the final form is done by a set of torch operations inside a dummy model without any trainable parameters. This helps ONNX convert those operations easily so that we don't have to write extra tensorflow code during inference to format our data. I have explained it in more detail over [here](https://www.kaggle.com/competitions/asl-signs/discussion/391301).\n\nIf you are looking to understand the data and task, you can check out my [EDA notebook](https://www.kaggle.com/code/mayukh18/sign-language-eda-visualization/).","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\n\nimport json\nfrom tqdm import tqdm\nimport numpy as np\nimport pandas as pd\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\n\nimport warnings\nwarnings.filterwarnings(action='ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-07T07:19:09.471791Z","iopub.execute_input":"2023-04-07T07:19:09.472371Z","iopub.status.idle":"2023-04-07T07:19:13.659369Z","shell.execute_reply.started":"2023-04-07T07:19:09.472336Z","shell.execute_reply":"2023-04-07T07:19:13.658126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LANDMARK_FILES_DIR = \"/kaggle/input/asl-signs/train_landmark_files\"\nTRAIN_FILE = \"/kaggle/input/asl-signs/train.csv\"\nlabel_map = json.load(open(\"/kaggle/input/asl-signs/sign_to_prediction_index_map.json\", \"r\"))","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:19:13.666051Z","iopub.execute_input":"2023-04-07T07:19:13.668821Z","iopub.status.idle":"2023-04-07T07:19:13.679766Z","shell.execute_reply.started":"2023-04-07T07:19:13.668774Z","shell.execute_reply":"2023-04-07T07:19:13.678634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Gen / Pre-Process Model\n\nConverts the (n_frames, 543, 3) data to (n_features,) form.","metadata":{}},{"cell_type":"code","source":"# class FeatureGen(nn.Module):\n#     def __init__(self):\n#         super(FeatureGen, self).__init__()\n#         pass\n    \n#     def forward(self, x):\n#         face_x = x[:,:468,:].contiguous().view(-1, 468*3)\n#         lefth_x = x[:,468:489,:].contiguous().view(-1, 21*3)\n#         pose_x = x[:,489:522,:].contiguous().view(-1, 33*3)\n#         righth_x = x[:,522:,:].contiguous().view(-1, 21*3)\n        \n#         lefth_x = lefth_x[~torch.any(torch.isnan(lefth_x), dim=1),:]\n#         righth_x = righth_x[~torch.any(torch.isnan(righth_x), dim=1),:]\n        \n#         x1m = torch.mean(face_x, 0)\n#         x2m = torch.mean(lefth_x, 0)\n#         x3m = torch.mean(pose_x, 0)\n#         x4m = torch.mean(righth_x, 0)\n        \n#         x1s = torch.std(face_x, 0)\n#         x2s = torch.std(lefth_x, 0)\n#         x3s = torch.std(pose_x, 0)\n#         x4s = torch.std(righth_x, 0)\n        \n#         xfeat = torch.cat([x1m,x2m,x3m,x4m, x1s,x2s,x3s,x4s], axis=0)\n#         xfeat = torch.where(torch.isnan(xfeat), torch.tensor(0.0, dtype=torch.float32), xfeat)\n        \n#         return xfeat\n# feature_converter = FeatureGen()\n\nclass FeatureGenSeq(nn.Module):\n    def __init__(self, max_length=220):\n        super().__init__()\n        self.max_length = max_length\n        self.LIP = [\n            61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n            291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n            78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n            95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n        ]\n        self.LHAND = list(range(468,489))\n        self.RHAND = list(range(522,543))\n        self.POSE = list(range(489,522))\n    \n    def pre_process(self, xyz):\n        xyz = xyz - xyz[~torch.isnan(xyz)].mean(0,keepdims=True) #noramlisation to common maen\n        xyz = xyz / xyz[~torch.isnan(xyz)].std(0, keepdims=True)\n        lip   = xyz[:, self.LIP]\n        lhand = xyz[:, self.LHAND]\n        # pose = xyz[:, self.POSE] #pose는 다음 링크의 InputNet에 포함되어 있지는 않음. https://www.kaggle.com/code/hengck23/lb-0-67-one-pytorch-transformer-solution\n        rhand = xyz[:, self.RHAND]\n        xyz = torch.cat([ #(none, 82, 3)\n            lip,\n            lhand,\n            # pose,\n            rhand,\n        ], axis = 1)\n        xyz[torch.isnan(xyz)] = 0\n        xyz = xyz[:self.max_length]\n        _, num_joints, joint_dim = xyz.shape\n        return xyz.reshape(_, num_joints * joint_dim)\n    \n    def forward(self, xyz):\n        frames,_,_ = xyz.shape\n        L = min([frames, self.max_length])\n        if isinstance(xyz, np.ndarray):\n            xyz = torch.tensor(xyz[:L,:,:])\n        return self.pre_process(xyz)\nfeature_converter = FeatureGenSeq(max_length=80)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:19:13.685190Z","iopub.execute_input":"2023-04-07T07:19:13.687973Z","iopub.status.idle":"2023-04-07T07:19:13.708378Z","shell.execute_reply.started":"2023-04-07T07:19:13.687927Z","shell.execute_reply":"2023-04-07T07:19:13.707008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Our actual model","metadata":{}},{"cell_type":"code","source":"# class ASLModel(nn.Module):\n#     def __init__(self, p):\n#         super(ASLModel, self).__init__()\n#         self.dropout = nn.Dropout(p)\n#         self.layer0 = nn.Linear(3258, 1024)\n#         self.layer1 = nn.Linear(1024, 512)\n#         self.layer2 = nn.Linear(512, 250)\n        \n#     def forward(self, x):\n#         x = self.layer0(x)\n#         x = self.dropout(x)\n#         x = self.layer1(x)\n#         x = self.layer2(x)\n#         return x","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:19:13.715809Z","iopub.execute_input":"2023-04-07T07:19:13.718319Z","iopub.status.idle":"2023-04-07T07:19:13.724890Z","shell.execute_reply.started":"2023-04-07T07:19:13.718270Z","shell.execute_reply":"2023-04-07T07:19:13.723689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def positional_encoding(length, embed_dim):\n    dim = embed_dim//2\n    position = np.arange(length)[:, np.newaxis]     # (seq, 1)\n    dim = np.arange(dim)[np.newaxis, :]/dim   # (1, dim)\n    angle = 1 / (10000**dim)         # (1, dim)\n    angle = position * angle    # (pos, dim)\n    pos_embed = np.concatenate(\n        [np.sin(angle), np.cos(angle)],\n        axis=-1\n    )\n    pos_embed = torch.from_numpy(pos_embed).float()\n    return pos_embed\n\n\nclass FeedForward(nn.Module):\n    def __init__(self, embed_dim, hidden_dim):\n        super().__init__()\n        self.mlp = nn.Sequential(\n            nn.Linear(embed_dim, hidden_dim),\n            nn.ReLU(inplace=True),\n            nn.Linear(hidden_dim, embed_dim),\n        )\n    def forward(self, x):\n        return self.mlp(x)\n\n\n#https://pytorch.org/docs/stable/generated/torch.nn.MultiheadAttention.html\nclass MultiHeadAttention(nn.Module):\n    def __init__(self,\n            embed_dim,\n            num_head,\n            batch_first,\n        ):\n        super().__init__()\n        self.mha = nn.MultiheadAttention(\n            embed_dim,\n            num_heads=num_head,\n            bias=True,\n            add_bias_kv=False,\n            kdim=None,\n            vdim=None,\n            dropout=0.0,\n            batch_first=batch_first,\n        )\n\n    def forward(self, x, x_mask):\n        out, _ = self.mha(x,x,x, key_padding_mask=x_mask)\n        return out\n\nclass TransformerBlock(nn.Module):\n    def __init__(self,\n        embed_dim,\n        num_head,\n        out_dim,\n        batch_first=True,\n    ):\n        super().__init__()\n        self.attn  = MultiHeadAttention(embed_dim, num_head,batch_first)\n        self.ffn   = FeedForward(embed_dim, out_dim)\n        self.norm1 = nn.LayerNorm(embed_dim)\n        self.norm2 = nn.LayerNorm(out_dim)\n\n    def forward(self, x, x_mask=None):\n        x = x + self.attn((self.norm1(x)), x_mask)\n        x = x + self.ffn((self.norm2(x)))\n        return x\n\n\nclass Net(nn.Module):\n    def __init__(\n            self,\n            num_class=250,\n            embed_dim=512,\n            max_length=220,\n            num_head=4,\n            num_layer=1,\n            p=0.4,\n            s_pose=True,\n            add_dxyz=False,\n            mean_pool=False,\n            output_type=['loss','inference'],\n            **kwargs\n        ):\n        super().__init__()\n        self.num_class = num_class\n        self.num_joint = 115 if s_pose else 82\n        self.max_length = max_length\n        self.num_head = num_head\n        self.num_layer = num_layer\n        self.p = p\n        self.output_type = output_type\n        self.joint_dim = 3*(1 + add_dxyz + mean_pool)\n\n        pos_embed = positional_encoding(self.max_length, embed_dim)\n        # self.register_buffer('pos_embed', pos_embed)\n        self.pos_embed = nn.Parameter(pos_embed)\n\n        self.cls_embed = nn.Parameter(torch.zeros((1, embed_dim)))\n        self.x_embed = nn.Sequential(\n            nn.Linear(self.num_joint * self.joint_dim, embed_dim, bias=False),\n        )\n\n        self.encoder = nn.ModuleList([\n            TransformerBlock(\n                embed_dim,\n                num_head,\n                embed_dim,\n            ) for _ in range(self.num_layer)\n        ])\n        self.logit = nn.Linear(embed_dim, num_class)\n\n    def forward(self, x=None, x_mask=None):\n        B,L,_ = x.shape\n        x = self.x_embed(x)\n        x = x + self.pos_embed[:L].unsqueeze(0)\n\n        x = torch.cat([\n            self.cls_embed.unsqueeze(0).repeat(B,1,1),\n            x\n        ],1)\n        if x_mask != None:\n            x_mask = torch.cat([\n                torch.zeros(B,1).to(x_mask),\n                x_mask\n            ],1)\n        \n        for block in self.encoder:\n            x = block(x,x_mask)\n            \n        cls = F.dropout(x[:,0], p = self.p, training=self.training)\n        logit = self.logit(cls)\n        return logit","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:19:13.727537Z","iopub.execute_input":"2023-04-07T07:19:13.728563Z","iopub.status.idle":"2023-04-07T07:19:13.762122Z","shell.execute_reply.started":"2023-04-07T07:19:13.728438Z","shell.execute_reply":"2023-04-07T07:19:13.760121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROWS_PER_FRAME = 543\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)\n\nclass ASLData(Dataset):\n    def __init__(self, datax, datay):\n        self.datax = datax\n        self.datay = datay\n        \n    def __getitem__(self, index):\n        return self.datax[index,:], self.datay[index]\n        \n    def __len__(self):\n        return len(self.datay)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:19:13.764200Z","iopub.execute_input":"2023-04-07T07:19:13.764602Z","iopub.status.idle":"2023-04-07T07:19:13.775647Z","shell.execute_reply.started":"2023-04-07T07:19:13.764554Z","shell.execute_reply":"2023-04-07T07:19:13.773741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Code for Feature Gen /Pre Process\nIt takes about 11 mins with multiprocessing, the current data is saved in a dataset. Run this code when you do your own processing.","metadata":{}},{"cell_type":"code","source":"import multiprocessing as mp\n\ndef convert_row(row):\n    x = load_relevant_data_subset(os.path.join(\"/kaggle/input/asl-signs\", row[1].path))\n    x = feature_converter(torch.tensor(x)).cpu().numpy()\n    return x, row[1].label\n\n# def convert_and_save_data():\n#     df = pd.read_csv(TRAIN_FILE)\n#     df['label'] = df['sign'].map(label_map)\n#     npdata = np.zeros((df.shape[0], 80, 246))\n#     nplabels = np.zeros(df.shape[0])\n#     with mp.Pool() as pool:\n#         results = pool.imap(convert_row, df.iterrows(), chunksize=250)\n#         for i, (x,y) in tqdm(enumerate(results), total=df.shape[0]):\n#             length = min([x.shape[0], 80])\n#             npdata[i,:length] = x[:length]\n#             nplabels[i] = y\n    \n#     np.save(\"feature_data.npy\", npdata)\n#     np.save(\"feature_labels.npy\", nplabels)\n        \n# convert_and_save_data()\n\n\ndf = pd.read_csv(TRAIN_FILE)\ndf = df.iloc[:1000]\ndf['label'] = df['sign'].map(label_map)\ndatax = np.zeros((df.shape[0], 80, 246))\ndatay = np.zeros(df.shape[0])\nwith mp.Pool() as pool:\n    results = pool.imap(convert_row, df.iterrows(), chunksize=250)\n    for i, (x,y) in tqdm(enumerate(results), total=df.shape[0]):\n        length = min([x.shape[0], 80])\n        datax[i,:length] = x[:length]\n        datay[i] = y","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:19:13.779164Z","iopub.execute_input":"2023-04-07T07:19:13.780515Z","iopub.status.idle":"2023-04-07T07:19:25.887757Z","shell.execute_reply.started":"2023-04-07T07:19:13.780477Z","shell.execute_reply":"2023-04-07T07:19:25.885474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"# datax = np.load(\"/kaggle/working/feature_data.npy\")\n# datay = np.load(\"/kaggle/working/feature_labels.npy\")","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:19:25.889577Z","iopub.execute_input":"2023-04-07T07:19:25.890236Z","iopub.status.idle":"2023-04-07T07:19:25.895935Z","shell.execute_reply.started":"2023-04-07T07:19:25.890192Z","shell.execute_reply":"2023-04-07T07:19:25.894660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS = 40\nBATCH_SIZE = 64\n\ntrainx, testx, trainy, testy = train_test_split(datax, datay, test_size=0.15, random_state=42)\n\ntrain_data = ASLData(trainx, trainy)\nvalid_data = ASLData(testx, testy)\n\ndef get_real_frame(x):\n        return np.count_nonzero(np.sum(np.sum(x,-1),-1))\n\ndef collate_fn(batch, mode=\"train\"):\n    num_frames = [get_real_frame(d[0]) for d in batch]\n    max_frame = max(num_frames)\n    batch_size = len(batch)\n    _, joints = batch[0][0].shape\n        \n    x = torch.zeros(size=(batch_size, max_frame, joints))\n    x_mask = torch.zeros(size=(batch_size,max_frame))\n    y = []\n    for i, d in enumerate(batch):\n        x_, y_ = d\n        x[i,:num_frames[i]] = torch.Tensor(x_[:num_frames[i],:])\n        x_mask[i,num_frames[i]:] = 1\n        y.append(y_)\n    \n    x_mask = x_mask > 0.5\n    inputs = {\n        \"x\" : x,\n        \"x_mask\" : x_mask if model==\"train\" else None,\n        \"y\" : torch.tensor(y).long()\n    }\n    return inputs\n\nfrom functools import partial\nfn = partial(collate_fn, mode='train')\ntrain_loader = DataLoader(train_data, collate_fn=fn, batch_size=BATCH_SIZE, num_workers=4, shuffle=True)\nfn = partial(collate_fn, mode='test')\nval_loader = DataLoader(valid_data, collate_fn=fn, batch_size=BATCH_SIZE, num_workers=4, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:19:25.898201Z","iopub.execute_input":"2023-04-07T07:19:25.898584Z","iopub.status.idle":"2023-04-07T07:19:25.988319Z","shell.execute_reply.started":"2023-04-07T07:19:25.898545Z","shell.execute_reply":"2023-04-07T07:19:25.987270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = ASLModel(0.2).cuda()\nmodel = Net(max_length=80,s_pose=False).to(\"cuda\")\nopt = torch.optim.Adam(model.parameters(), lr=0.005)\ncriterion = nn.CrossEntropyLoss()\nsched = torch.optim.lr_scheduler.StepLR(opt, step_size=300, gamma=0.95)\n\nfor i in range(EPOCHS):\n    model.train()\n    \n    train_loss_sum = 0.\n    train_correct = 0\n    train_total = 0\n    train_bar = train_loader\n    for batch in train_bar:\n        y = batch.pop(\"y\")\n        batch = {k:torch.Tensor(v).float().cuda() for k,v in batch.items() if v != None}\n        y = torch.Tensor(y).long().cuda()  \n        y_pred = model(**batch)\n        \n        loss = criterion(y_pred, y)\n        loss.backward()\n        opt.step()\n        opt.zero_grad()\n        \n        train_loss_sum += loss.item()\n        train_correct += np.sum((np.argmax(y_pred.detach().cpu().numpy(), axis=1) == y.cpu().numpy()))\n        train_total += 1\n        sched.step()\n        \n    val_loss_sum = 0.\n    val_correct = 0\n    val_total = 0\n    model.eval()\n    for batch in val_loader:\n        y = batch.pop(\"y\")\n        batch = {k:torch.Tensor(v).float().cuda() for k,v in batch.items() if v != None}\n        y = torch.Tensor(y).long().cuda()\n        \n        with torch.no_grad():\n            y_pred = model(**batch)\n            loss = criterion(y_pred, y)\n            val_loss_sum += loss.item()\n            val_correct += np.sum((np.argmax(y_pred.cpu().numpy(), axis=1) == y.cpu().numpy()))\n            val_total += 1\n                              \n    print(f\"Epoch:{i} > Train Loss: {(train_loss_sum/train_total):.04f}, Train Acc: {train_correct/len(train_data):0.04f}\")\n    print(f\"Epoch:{i} > Val Loss: {(val_loss_sum/val_total):.04f}, Val Acc: {val_correct/len(valid_data):0.04f}\")\n    print(\"=\"*50)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:19:25.992710Z","iopub.execute_input":"2023-04-07T07:19:25.993029Z","iopub.status.idle":"2023-04-07T07:19:51.396171Z","shell.execute_reply.started":"2023-04-07T07:19:25.992998Z","shell.execute_reply":"2023-04-07T07:19:51.394868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:19:51.398260Z","iopub.execute_input":"2023-04-07T07:19:51.398662Z","iopub.status.idle":"2023-04-07T07:19:51.577609Z","shell.execute_reply.started":"2023-04-07T07:19:51.398617Z","shell.execute_reply":"2023-04-07T07:19:51.576446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tensorflow Conversion","metadata":{}},{"cell_type":"code","source":"!pip install onnx-tf\n!pip install tflite-runtime","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:19:51.579272Z","iopub.execute_input":"2023-04-07T07:19:51.579974Z","iopub.status.idle":"2023-04-07T07:20:13.993069Z","shell.execute_reply.started":"2023-04-07T07:19:51.579930Z","shell.execute_reply":"2023-04-07T07:20:13.991810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def positional_encoding(length, embed_dim):\n    dim = embed_dim//2\n    position = np.arange(length)[:, np.newaxis]     # (seq, 1)\n    dim = np.arange(dim)[np.newaxis, :]/dim   # (1, dim)\n    angle = 1 / (10000**dim)         # (1, dim)\n    angle = position * angle    # (pos, dim)\n    pos_embed = np.concatenate(\n        [np.sin(angle), np.cos(angle)],\n        axis=-1\n    )\n    pos_embed = torch.from_numpy(pos_embed).float()\n    return pos_embed\n\n\nclass FeedForward(nn.Module):\n    def __init__(self, embed_dim, hidden_dim):\n        super().__init__()\n        self.mlp = nn.Sequential(\n            nn.Linear(embed_dim, hidden_dim),\n            nn.ReLU(inplace=True),\n            nn.Linear(hidden_dim, embed_dim),\n        )\n    def forward(self, x):\n        return self.mlp(x)\n\n\n#https://pytorch.org/docs/stable/generated/torch.nn.MultiheadAttention.html\nclass MultiHeadAttention(nn.Module):\n    def __init__(self,\n            embed_dim,\n            num_head,\n            batch_first,\n        ):\n        super().__init__()\n        self.mha = nn.MultiheadAttention(\n            embed_dim,\n            num_heads=num_head,\n            bias=True,\n            add_bias_kv=False,\n            kdim=None,\n            vdim=None,\n            dropout=0.0,\n            batch_first=batch_first,\n        )\n\n    def forward(self, x, x_mask):\n        out, _ = self.mha(x,x,x, key_padding_mask=x_mask)\n        return out\n\nclass TransformerBlock(nn.Module):\n    def __init__(self,\n        embed_dim,\n        num_head,\n        out_dim,\n        batch_first=True,\n    ):\n        super().__init__()\n        self.attn  = MultiHeadAttention(embed_dim, num_head,batch_first)\n        self.ffn   = FeedForward(embed_dim, out_dim)\n        self.norm1 = nn.LayerNorm(embed_dim)\n        self.norm2 = nn.LayerNorm(out_dim)\n\n    def forward(self, x, x_mask=None):\n        x = x + self.attn((self.norm1(x)), x_mask)\n        x = x + self.ffn((self.norm2(x)))\n        return x\n\n\nclass Net(nn.Module):\n    def __init__(\n            self,\n            num_class=250,\n            embed_dim=512,\n            max_length=220,\n            num_head=4,\n            num_layer=1,\n            p=0.4,\n            s_pose=True,\n            add_dxyz=False,\n            mean_pool=False,\n            output_type=['loss','inference'],\n            **kwargs\n        ):\n        super().__init__()\n        self.num_class = num_class\n        self.num_joint = 115 if s_pose else 82\n        self.max_length = max_length\n        self.num_head = num_head\n        self.num_layer = num_layer\n        self.p = p\n        self.output_type = output_type\n        self.joint_dim = 3*(1 + add_dxyz + mean_pool)\n\n        pos_embed = positional_encoding(self.max_length, embed_dim)\n        # self.register_buffer('pos_embed', pos_embed)\n        self.pos_embed = nn.Parameter(pos_embed)\n\n        self.cls_embed = nn.Parameter(torch.zeros((1, embed_dim)))\n        self.x_embed = nn.Sequential(\n            nn.Linear(self.num_joint * self.joint_dim, embed_dim, bias=False),\n        )\n\n        self.encoder = nn.ModuleList([\n            TransformerBlock(\n                embed_dim,\n                num_head,\n                embed_dim,\n            ) for _ in range(self.num_layer)\n        ])\n        self.logit = nn.Linear(embed_dim, num_class)\n\n    def forward(self, x):\n        x = torch.where(torch.isnan(x), torch.tensor(0.0, dtype=torch.float32).cuda(), x)\n        B,L,_ = x.shape\n        x = self.x_embed(x)\n        x = x + self.pos_embed[:L].unsqueeze(0)\n\n        x = torch.cat([\n            self.cls_embed.unsqueeze(0).repeat(B,1,1),\n            x\n        ],1)\n        \n        for block in self.encoder:\n            x = block(x = x)\n            \n        cls = F.dropout(x[:,0], p = self.p, training=self.training)\n        logit = self.logit(cls)\n        return logit","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:23:11.896731Z","iopub.execute_input":"2023-04-07T07:23:11.897186Z","iopub.status.idle":"2023-04-07T07:23:11.919353Z","shell.execute_reply.started":"2023-04-07T07:23:11.897126Z","shell.execute_reply":"2023-04-07T07:23:11.918216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_infe = Net(max_length=80,s_pose=False).cuda()\nmodel_infe.load_state_dict(model.state_dict())","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:23:13.037447Z","iopub.execute_input":"2023-04-07T07:23:13.038161Z","iopub.status.idle":"2023-04-07T07:23:13.071024Z","shell.execute_reply.started":"2023-04-07T07:23:13.038107Z","shell.execute_reply":"2023-04-07T07:23:13.069669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_input = torch.rand((50, 543, 3))\nonnx_feat_gen_path = 'feature_gen.onnx'\n\nfeature_converter.eval()\n\ntorch.onnx.export(\n    feature_converter,                  # PyTorch Model\n    sample_input,                    # Input tensor\n    onnx_feat_gen_path,        # Output file (eg. 'output_model.onnx')\n    opset_version=12,       # Operator support version\n    input_names=['input'],   # Input tensor name (arbitary)\n    output_names=['output'], # Output tensor name (arbitary)\n    dynamic_axes={\n        'input' : {0: 'input'}\n    }\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:23:13.627035Z","iopub.execute_input":"2023-04-07T07:23:13.627724Z","iopub.status.idle":"2023-04-07T07:23:13.672966Z","shell.execute_reply.started":"2023-04-07T07:23:13.627685Z","shell.execute_reply":"2023-04-07T07:23:13.671884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_input = torch.rand((1, 1, 246)).cuda()\nonnx_model_path = 'asl_model.onnx'\n\nmodel.eval()\n\ntorch.onnx.export(\n    model_infe,                  # PyTorch Model\n    sample_input,                    # Input tensor\n    onnx_model_path,        # Output file (eg. 'output_model.onnx')\n    opset_version=12,       # Operator support version\n    input_names=['input'],   # Input tensor name (arbitary)\n    output_names=['output'], # Output tensor name (arbitary)\n    dynamic_axes={\n        'input' : {0: 'input'}\n    }\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:23:14.258354Z","iopub.execute_input":"2023-04-07T07:23:14.259294Z","iopub.status.idle":"2023-04-07T07:23:14.418127Z","shell.execute_reply.started":"2023-04-07T07:23:14.259257Z","shell.execute_reply":"2023-04-07T07:23:14.417072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import onnx\nfrom onnx_tf.backend import prepare\n\n\ntf_feat_gen_path = '/kaggle/working/feature_gen'\nonnx_feat_gen = onnx.load(onnx_feat_gen_path)\ntf_rep = prepare(onnx_feat_gen)\ntf_rep.export_graph(tf_feat_gen_path)\n\n\ntf_model_path = '/kaggle/working/asl_model'\nonnx_model = onnx.load(onnx_model_path)\ntf_rep = prepare(onnx_model)\ntf_rep.export_graph(tf_model_path)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:23:15.138487Z","iopub.execute_input":"2023-04-07T07:23:15.139426Z","iopub.status.idle":"2023-04-07T07:23:36.927175Z","shell.execute_reply.started":"2023-04-07T07:23:15.139360Z","shell.execute_reply":"2023-04-07T07:23:36.925925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Final Inference Model in Tensorflow\nBoth of the converted models will be used here one after another.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n\nclass ASLInferModel(tf.Module):\n    def __init__(self):\n        super(ASLInferModel, self).__init__()\n        self.feature_gen = tf.saved_model.load(tf_feat_gen_path)\n        self.model = tf.saved_model.load(tf_model_path)\n        self.feature_gen.trainable = False\n        self.model.trainable = False\n    \n    @tf.function(input_signature=[\n      tf.TensorSpec(shape=[None, 543, 3], dtype=tf.float32, name='inputs')\n    ])\n    def call(self, input):\n        output_tensors = {}\n        features = self.feature_gen(**{'input': input})['output']\n        output_tensors['outputs'] = self.model(**{'input': tf.expand_dims(features, 0)})['output'][0,:]\n        return output_tensors\n    \n    \nmytfmodel = ASLInferModel()\ntf.saved_model.save(mytfmodel, '/kaggle/working/tf_infer_model', signatures={'serving_default': mytfmodel.call})","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:23:36.929882Z","iopub.execute_input":"2023-04-07T07:23:36.930316Z","iopub.status.idle":"2023-04-07T07:23:39.233096Z","shell.execute_reply.started":"2023-04-07T07:23:36.930275Z","shell.execute_reply":"2023-04-07T07:23:39.231997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# Convert the model\n\ntf_infer_model_path = '/kaggle/working/tf_infer_model'\nconverter = tf.lite.TFLiteConverter.from_saved_model(tf_infer_model_path)\nconverter.target_spec.supported_ops = [\n  tf.lite.OpsSet.TFLITE_BUILTINS, # enable TensorFlow Lite ops.\n  tf.lite.OpsSet.SELECT_TF_OPS # enable TensorFlow ops.\n]\ntflite_model = converter.convert()\n\ntflite_model_path = 'model.tflite'\n\n# Save the model\nwith open(tflite_model_path, 'wb') as f:\n    f.write(tflite_model)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:23:39.234854Z","iopub.execute_input":"2023-04-07T07:23:39.235261Z","iopub.status.idle":"2023-04-07T07:23:42.254877Z","shell.execute_reply.started":"2023-04-07T07:23:39.235217Z","shell.execute_reply":"2023-04-07T07:23:42.253572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROWS_PER_FRAME = 543  # number of landmarks per frame\npq_path = \"/kaggle/input/asl-signs/train_landmark_files/53618/1001379621.parquet\"\n\nimport tflite_runtime.interpreter as tflite\ninterpreter = tflite.Interpreter(tflite_model_path)\ninterpreter.allocate_tensors()\n\nfound_signatures = list(interpreter.get_signature_list().keys())\n\n# if REQUIRED_SIGNATURE not in found_signatures:\n#     raise KernelEvalException('Required input signature not found.')\n\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")\noutput = prediction_fn(inputs=load_relevant_data_subset(pq_path))\nsign = np.argmax(output[\"outputs\"])\n\nprint(sign, output[\"outputs\"].shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:23:42.257632Z","iopub.execute_input":"2023-04-07T07:23:42.258043Z","iopub.status.idle":"2023-04-07T07:23:42.316793Z","shell.execute_reply.started":"2023-04-07T07:23:42.258001Z","shell.execute_reply":"2023-04-07T07:23:42.315606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission.zip $tflite_model_path","metadata":{"execution":{"iopub.status.busy":"2023-04-07T07:23:42.318491Z","iopub.execute_input":"2023-04-07T07:23:42.319588Z","iopub.status.idle":"2023-04-07T07:23:43.998962Z","shell.execute_reply.started":"2023-04-07T07:23:42.319545Z","shell.execute_reply":"2023-04-07T07:23:43.997650Z"},"trusted":true},"execution_count":null,"outputs":[]}]}