{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":46105,"databundleVersionId":5087314,"sourceType":"competition"},{"sourceId":7549352,"sourceType":"datasetVersion","datasetId":4396741},{"sourceId":7549849,"sourceType":"datasetVersion","datasetId":4397074},{"sourceId":2632847,"sourceType":"datasetVersion","datasetId":1589971}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-07T04:34:57.822574Z","iopub.execute_input":"2024-02-07T04:34:57.823594Z","iopub.status.idle":"2024-02-07T04:34:59.102948Z","shell.execute_reply.started":"2024-02-07T04:34:57.823550Z","shell.execute_reply":"2024-02-07T04:34:59.101733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install tensorflow-addons --quiet","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:34:59.105239Z","iopub.execute_input":"2024-02-07T04:34:59.106082Z","iopub.status.idle":"2024-02-07T04:35:16.665635Z","shell.execute_reply.started":"2024-02-07T04:34:59.106042Z","shell.execute_reply":"2024-02-07T04:35:16.664265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, json, random, math, scipy\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import StratifiedGroupKFold \nfrom types import SimpleNamespace\nfrom pathlib import Path","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:16.667758Z","iopub.execute_input":"2024-02-07T04:35:16.668128Z","iopub.status.idle":"2024-02-07T04:35:34.043407Z","shell.execute_reply.started":"2024-02-07T04:35:16.668095Z","shell.execute_reply":"2024-02-07T04:35:34.042459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:37:32.595096Z","iopub.execute_input":"2024-02-07T04:37:32.595555Z","iopub.status.idle":"2024-02-07T04:37:32.603106Z","shell.execute_reply.started":"2024-02-07T04:37:32.595522Z","shell.execute_reply":"2024-02-07T04:37:32.601765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg = SimpleNamespace()\niskaggle = os.environ.get('KAGGLE_KERNEL_RUN_TYPE', '')\ncfg.PREPROCESS_DATA = False\ncfg.TRAIN_MODEL = True\ncfg.N_ROWS = 543\ncfg.N_DIMS = 3\ncfg.DIM_NAMES = ['x', 'y', 'z']\ncfg.SEED = 42\ncfg.NUM_CLASSES = 250\ncfg.IS_INTERACTIVE = True\ncfg.VERBOSE = 2\ncfg.INPUT_SIZE = 32\ncfg.BATCH_ALL_SIGNS_N = 4\ncfg.BATCH_SIZE = 256\ncfg.N_EPOCHS = 50\ncfg.LR_MAX = 1e-3\ncfg.N_WARMUP_EPOCHS = 0\ncfg.WD_RATIO = 0.05\ncfg.MASK_VAL = 4237","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.056237Z","iopub.execute_input":"2024-02-07T04:35:34.056555Z","iopub.status.idle":"2024-02-07T04:35:34.080385Z","shell.execute_reply.started":"2024-02-07T04:35:34.056528Z","shell.execute_reply":"2024-02-07T04:35:34.079187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# landmark indices in original data\nLIPS_IDXS0 = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\nLEFT_HAND_IDXS0  = np.arange(468,489)\nRIGHT_HAND_IDXS0 = np.arange(522,543)\nPOSE_IDXS0       = np.arange(502, 512)\nLANDMARK_IDXS0   = np.concatenate((LIPS_IDXS0, LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0, POSE_IDXS0))\nHAND_IDXS0       = np.concatenate((LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0), axis=0)\nN_COLS           = LANDMARK_IDXS0.size","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.081909Z","iopub.execute_input":"2024-02-07T04:35:34.082278Z","iopub.status.idle":"2024-02-07T04:35:34.096070Z","shell.execute_reply.started":"2024-02-07T04:35:34.082244Z","shell.execute_reply":"2024-02-07T04:35:34.094722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Landmark indices in processed data\nLIPS_IDXS       = np.argwhere(np.isin(LANDMARK_IDXS0, LIPS_IDXS0)).squeeze()\nLEFT_HAND_IDXS  = np.argwhere(np.isin(LANDMARK_IDXS0, LEFT_HAND_IDXS0)).squeeze()\nRIGHT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS0, RIGHT_HAND_IDXS0)).squeeze()\nHAND_IDXS       = np.argwhere(np.isin(LANDMARK_IDXS0, HAND_IDXS0)).squeeze()\nPOSE_IDXS       = np.argwhere(np.isin(LANDMARK_IDXS0, POSE_IDXS0)).squeeze()","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.097439Z","iopub.execute_input":"2024-02-07T04:35:34.098112Z","iopub.status.idle":"2024-02-07T04:35:34.112878Z","shell.execute_reply.started":"2024-02-07T04:35:34.098073Z","shell.execute_reply":"2024-02-07T04:35:34.111828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LIPS_START = 0\nLEFT_HAND_START = LIPS_IDXS.size\nRIGHT_HAND_START = LEFT_HAND_START + LEFT_HAND_IDXS.size\nPOSE_START = RIGHT_HAND_START + RIGHT_HAND_IDXS.size","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.114134Z","iopub.execute_input":"2024-02-07T04:35:34.115196Z","iopub.status.idle":"2024-02-07T04:35:34.122191Z","shell.execute_reply.started":"2024-02-07T04:35:34.115077Z","shell.execute_reply":"2024-02-07T04:35:34.120245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LIPS_MEAN = np.load('/kaggle/input/np-saves/lips_mean.npy')\nLIPS_STD = np.load('/kaggle/input/np-saves/lips_std.npy')","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.123670Z","iopub.execute_input":"2024-02-07T04:35:34.123985Z","iopub.status.idle":"2024-02-07T04:35:34.140487Z","shell.execute_reply.started":"2024-02-07T04:35:34.123959Z","shell.execute_reply":"2024-02-07T04:35:34.139199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LEFT_HANDS_MEAN = np.load('/kaggle/input/np-saves/lh_mean.npy')\nLEFT_HANDS_STD = np.load('/kaggle/input/np-saves/lh_std.npy')\nRIGHT_HANDS_MEAN = np.load('/kaggle/input/np-saves/rh_mean.npy')\nRIGHT_HANDS_STD = np.load('/kaggle/input/np-saves/rh_std.npy')","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.144658Z","iopub.execute_input":"2024-02-07T04:35:34.145347Z","iopub.status.idle":"2024-02-07T04:35:34.165664Z","shell.execute_reply.started":"2024-02-07T04:35:34.145316Z","shell.execute_reply":"2024-02-07T04:35:34.164665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"POSE_MEAN = np.load('/kaggle/input/np-saves/pose_mean.npy')\nPOSE_STD = np.load('/kaggle/input/np-saves/pose_std.npy')","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.167041Z","iopub.execute_input":"2024-02-07T04:35:34.168102Z","iopub.status.idle":"2024-02-07T04:35:34.180103Z","shell.execute_reply.started":"2024-02-07T04:35:34.168059Z","shell.execute_reply":"2024-02-07T04:35:34.179045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# function to crunch the entire video to 32 non empty frames and each frame only contains 92 indices (considering only lips in entire face)\nclass PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__()\n        \n    def pad_edge(self, t, repeats, side):\n        if side == 'LEFT':\n            return tf.concat((tf.repeat(t[:1], repeats=repeats, axis=0), t), axis=0)\n        elif side == 'RIGHT':\n            return tf.concat((t, tf.repeat(t[-1:], repeats=repeats, axis=0)), axis=0)\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,cfg.N_ROWS,cfg.N_DIMS], dtype=tf.float32),),\n    )\n    def call(self, data0):\n        # Number of Frames in Video\n        N_FRAMES0 = tf.shape(data0)[0]\n        \n        # Keep only non-empty frames in data\n        frames_hands_nansum = tf.experimental.numpy.nanmean(tf.gather(data0, HAND_IDXS0, axis=1), axis=[1,2])\n        non_empty_frames_idxs = tf.where(frames_hands_nansum > 0)\n        non_empty_frames_idxs = tf.squeeze(non_empty_frames_idxs, axis=1)\n        data = tf.gather(data0, non_empty_frames_idxs, axis=0)\n        \n        non_empty_frames_idxs = tf.cast(non_empty_frames_idxs, tf.float32) \n        \n        # Number of non-empty frames\n        N_FRAMES = tf.shape(data)[0]\n        data = tf.gather(data, LANDMARK_IDXS0, axis=1)\n        \n        if N_FRAMES < cfg.INPUT_SIZE:\n            # Video fits in cfg.INPUT_SIZE\n            non_empty_frames_idxs = tf.pad(non_empty_frames_idxs, [[0, cfg.INPUT_SIZE-N_FRAMES]], constant_values=-1)\n            data = tf.pad(data, [[0, cfg.INPUT_SIZE-N_FRAMES], [0,0], [0,0]], constant_values=0)\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            return data, non_empty_frames_idxs\n        else:\n            # Video needs to be downsampled to cfg.INPUT_SIZE\n            if N_FRAMES < cfg.INPUT_SIZE**2:\n                repeats = tf.math.floordiv(cfg.INPUT_SIZE * cfg.INPUT_SIZE, N_FRAMES0)\n                data = tf.repeat(data, repeats=repeats, axis=0)\n                non_empty_frames_idxs = tf.repeat(non_empty_frames_idxs, repeats=repeats, axis=0)\n\n            # Pad To Multiple Of Input Size\n            pool_size = tf.math.floordiv(len(data), cfg.INPUT_SIZE)\n            if tf.math.mod(len(data), cfg.INPUT_SIZE) > 0:\n                pool_size += 1\n            if pool_size == 1:\n                pad_size = (pool_size * cfg.INPUT_SIZE) - len(data)\n            else:\n                pad_size = (pool_size * cfg.INPUT_SIZE) % len(data)\n\n            # Pad Start/End with Start/End value\n            pad_left = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(cfg.INPUT_SIZE, 2)\n            pad_right = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(cfg.INPUT_SIZE, 2)\n            if tf.math.mod(pad_size, 2) > 0:\n                pad_right += 1\n\n            # Pad By Concatenating Left/Right Edge Values\n            data = self.pad_edge(data, pad_left, 'LEFT')\n            data = self.pad_edge(data, pad_right, 'RIGHT')\n\n            # Pad Non Empty Frame Indices\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_left, 'LEFT')\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_right, 'RIGHT')\n\n            # Reshape to Mean Pool\n            data = tf.reshape(data, [cfg.INPUT_SIZE, -1, N_COLS, cfg.N_DIMS])\n            non_empty_frames_idxs = tf.reshape(non_empty_frames_idxs, [cfg.INPUT_SIZE, -1])\n\n            # Mean Pool\n            data = tf.experimental.numpy.nanmean(data, axis=1)\n            non_empty_frames_idxs = tf.experimental.numpy.nanmean(non_empty_frames_idxs, axis=1)\n\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            \n            return data, non_empty_frames_idxs\n    \npreprocess_layer = PreprocessLayer()","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.181651Z","iopub.execute_input":"2024-02-07T04:35:34.182465Z","iopub.status.idle":"2024-02-07T04:35:34.233183Z","shell.execute_reply.started":"2024-02-07T04:35:34.182431Z","shell.execute_reply":"2024-02-07T04:35:34.232120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def scaled_dot_product(q,k,v, softmax, attention_mask):\n    #calculates Q . K(transpose)\n    qkt = tf.matmul(q,k,transpose_b=True)\n    #caculates scaling factor\n    dk = tf.math.sqrt(tf.cast(q.shape[-1],dtype=tf.float32))\n    scaled_qkt = qkt/dk\n    softmax = softmax(scaled_qkt, mask=attention_mask)\n    \n    z = tf.matmul(softmax,v)\n    #shape: (m,Tx,depth), same shape as q,k,v\n    return z\n\nclass MultiHeadAttention(tf.keras.layers.Layer):\n    def __init__(self,d_model,num_of_heads):\n        super(MultiHeadAttention,self).__init__()\n        self.d_model = d_model\n        self.num_of_heads = num_of_heads\n        self.depth = d_model//num_of_heads\n        self.wq = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wk = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wv = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wo = tf.keras.layers.Dense(d_model)\n        self.softmax = tf.keras.layers.Softmax()\n        \n    def call(self,x, attention_mask):\n        \n        multi_attn = []\n        for i in range(self.num_of_heads):\n            Q = self.wq[i](x)\n            K = self.wk[i](x)\n            V = self.wv[i](x)\n            multi_attn.append(scaled_dot_product(Q,K,V, self.softmax, attention_mask))\n            \n        multi_head = tf.concat(multi_attn,axis=-1)\n        multi_head_attention = self.wo(multi_head)\n        return multi_head_attention","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.235115Z","iopub.execute_input":"2024-02-07T04:35:34.236029Z","iopub.status.idle":"2024-02-07T04:35:34.250417Z","shell.execute_reply.started":"2024-02-07T04:35:34.235995Z","shell.execute_reply":"2024-02-07T04:35:34.249272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Epsilon value for layer normalisation\nLAYER_NORM_EPS = 1e-6\n\n# Dense layer units for landmarks\nLIPS_UNITS = 384\nHANDS_UNITS = 384\nPOSE_UNITS = 384\n# final embedding and transformer embedding size\nUNITS = 384\n\n# Transformer\nNUM_BLOCKS = 2\nMLP_RATIO = 2\n\n# Dropout\nEMBEDDING_DROPOUT = 0.00\nMLP_DROPOUT_RATIO = 0.30\nCLASSIFIER_DROPOUT_RATIO = 0.10\n\n# Initiailizers\nINIT_HE_UNIFORM = tf.keras.initializers.he_uniform\nINIT_GLOROT_UNIFORM = tf.keras.initializers.glorot_uniform\nINIT_ZEROS = tf.keras.initializers.constant(0.0)\n# Activations\nGELU = tf.keras.activations.gelu","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.252095Z","iopub.execute_input":"2024-02-07T04:35:34.252488Z","iopub.status.idle":"2024-02-07T04:35:34.265956Z","shell.execute_reply.started":"2024-02-07T04:35:34.252457Z","shell.execute_reply":"2024-02-07T04:35:34.264844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Transformer(tf.keras.Model):\n    def __init__(self, num_blocks):\n        super(Transformer, self).__init__(name='transformer')\n        self.num_blocks = num_blocks\n    \n    def build(self, input_shape):\n        self.ln_1s = []\n        self.mhas = []\n        self.ln_2s = []\n        self.mlps = []\n        # Make Transformer Blocks\n        for i in range(self.num_blocks):\n            # First Layer Normalisation\n            self.ln_1s.append(tf.keras.layers.LayerNormalization(epsilon=LAYER_NORM_EPS))\n            # Multi Head Attention\n            self.mhas.append(MultiHeadAttention(UNITS, 8))\n            # Second Layer Normalisation\n            self.ln_2s.append(tf.keras.layers.LayerNormalization(epsilon=LAYER_NORM_EPS))\n            # Multi Layer Perception\n            self.mlps.append(tf.keras.Sequential([\n                tf.keras.layers.Dense(UNITS * MLP_RATIO, activation=GELU, kernel_initializer=INIT_GLOROT_UNIFORM),\n                tf.keras.layers.Dropout(MLP_DROPOUT_RATIO),\n                tf.keras.layers.Dense(UNITS, kernel_initializer=INIT_HE_UNIFORM),\n            ]))\n        \n    def call(self, x, attention_mask):\n        # Iterate input over transformer blocks\n        for ln_1, mha, ln_2, mlp in zip(self.ln_1s, self.mhas, self.ln_2s, self.mlps):\n            x1 = ln_1(x)\n            attention_output = mha(x1, attention_mask)\n            x2 = x1 + attention_output\n            x3 = ln_2(x2)\n            x3 = mlp(x3)\n            x = x3 + x2\n    \n        return x","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.267851Z","iopub.execute_input":"2024-02-07T04:35:34.268261Z","iopub.status.idle":"2024-02-07T04:35:34.281358Z","shell.execute_reply.started":"2024-02-07T04:35:34.268231Z","shell.execute_reply":"2024-02-07T04:35:34.279913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LandmarkEmbedding(tf.keras.Model):\n    def __init__(self, units, name):\n        super(LandmarkEmbedding, self).__init__(name=f'{name}_embedding')\n        self.units = units\n        \n    def build(self, input_shape):\n        # Embedding for missing landmark in frame, initizlied with zeros\n        self.empty_embedding = self.add_weight(\n            name=f'{self.name}_empty_embedding',\n            shape=[self.units],\n            initializer=INIT_ZEROS,\n        )\n        # Embedding\n        self.dense = tf.keras.Sequential([\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM, activation=GELU),\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name=f'{self.name}_dense')\n\n    def call(self, x):\n        return tf.where(\n                # Checks whether landmark is missing in frame\n                tf.reduce_sum(x, axis=2, keepdims=True) == 0,\n                # If so, the empty embedding is used\n                self.empty_embedding,\n                # Otherwise the landmark data is embedded\n                self.dense(x),\n            )","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.282953Z","iopub.execute_input":"2024-02-07T04:35:34.283303Z","iopub.status.idle":"2024-02-07T04:35:34.298606Z","shell.execute_reply.started":"2024-02-07T04:35:34.283273Z","shell.execute_reply":"2024-02-07T04:35:34.297488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CustomEmbedding(tf.keras.Model):\n    def __init__(self):\n        super(CustomEmbedding, self).__init__()\n        \n    def get_diffs(self, l):\n        S = l.shape[2]\n        other = tf.expand_dims(l, 3)\n        other = tf.repeat(other, S, axis=3)\n        other = tf.transpose(other, [0,1,3,2])\n        diffs = tf.expand_dims(l, 3) - other\n        diffs = tf.reshape(diffs, [-1, cfg.INPUT_SIZE, S*S])\n        return diffs\n\n    def build(self, input_shape):\n        # Positional Embedding, initialized with zeros\n        self.positional_embedding = tf.keras.layers.Embedding(cfg.INPUT_SIZE+1, UNITS, embeddings_initializer=INIT_ZEROS)\n        # Embedding layer for Landmarks\n        self.lips_embedding = LandmarkEmbedding(LIPS_UNITS, 'lips')\n        self.left_hand_embedding = LandmarkEmbedding(HANDS_UNITS, 'left_hand')\n        self.right_hand_embedding = LandmarkEmbedding(HANDS_UNITS, 'right_hand')\n        self.pose_embedding = LandmarkEmbedding(POSE_UNITS, 'pose')\n        # Landmark Weights\n        self.landmark_weights = tf.Variable(tf.zeros([4], dtype=tf.float32), name='landmark_weights')\n        # Fully Connected Layers for combined landmarks\n        self.fc = tf.keras.Sequential([\n            tf.keras.layers.Dense(UNITS, name='fully_connected_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM, activation=GELU),\n            tf.keras.layers.Dense(UNITS, name='fully_connected_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name='fc')\n\n\n    def call(self, lips0, left_hand0, right_hand0, pose0, non_empty_frame_idxs, training=False):\n        # Lips\n        lips_embedding = self.lips_embedding(lips0)\n        # Left Hand\n        left_hand_embedding = self.left_hand_embedding(left_hand0)\n        # Right Hand\n        right_hand_embedding = self.right_hand_embedding(right_hand0)\n        # Pose\n        pose_embedding = self.pose_embedding(pose0)\n        # Merge Embeddings of all landmarks with mean pooling\n        x = tf.stack((lips_embedding, left_hand_embedding, right_hand_embedding, pose_embedding), axis=3)\n        # Merge Landmarks with trainable attention weights\n        x = x * tf.nn.softmax(self.landmark_weights)\n        x = tf.reduce_sum(x, axis=3)\n        # Fully Connected Layers\n        x = self.fc(x)\n        # Add Positional Embedding\n        normalised_non_empty_frame_idxs = tf.where(\n            tf.math.equal(non_empty_frame_idxs, -1.0),\n            cfg.INPUT_SIZE,\n            tf.cast(\n                non_empty_frame_idxs / tf.reduce_max(non_empty_frame_idxs, axis=1, keepdims=True) * cfg.INPUT_SIZE,\n                tf.int32,\n            ),\n        )\n        x = x + self.positional_embedding(normalised_non_empty_frame_idxs)\n        \n        return x","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.300438Z","iopub.execute_input":"2024-02-07T04:35:34.300869Z","iopub.status.idle":"2024-02-07T04:35:34.318945Z","shell.execute_reply.started":"2024-02-07T04:35:34.300829Z","shell.execute_reply":"2024-02-07T04:35:34.317822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_lr_metric(optimizer):\n    def lr(y_true, y_pred):\n        return optimizer.lr\n    return lr","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.320982Z","iopub.execute_input":"2024-02-07T04:35:34.321420Z","iopub.status.idle":"2024-02-07T04:35:34.335327Z","shell.execute_reply.started":"2024-02-07T04:35:34.321382Z","shell.execute_reply":"2024-02-07T04:35:34.333864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    # Inputs\n    frames = tf.keras.layers.Input([cfg.INPUT_SIZE, N_COLS, cfg.N_DIMS], dtype=tf.float32, name='frames')\n    non_empty_frame_idxs = tf.keras.layers.Input([cfg.INPUT_SIZE], dtype=tf.float32, name='non_empty_frame_idxs')\n    # Padding Mask\n    mask = tf.cast(tf.math.not_equal(non_empty_frame_idxs, -1), tf.float32)\n    mask = tf.expand_dims(mask, axis=2)\n    \n    x = frames\n    x = tf.slice(x, [0,0,0,0], [-1,cfg.INPUT_SIZE, N_COLS, 2])\n    # LIPS\n    lips = tf.slice(x, [0,0,LIPS_START,0], [-1,cfg.INPUT_SIZE, 40, 2])\n    lips = tf.where(\n            tf.math.equal(lips, 0.0),\n            0.0,\n            (lips - LIPS_MEAN) / LIPS_STD,\n        )\n    lips = tf.reshape(lips, [-1, cfg.INPUT_SIZE, 40*2])\n    # LEFT HAND\n    left_hand = tf.slice(x, [0,0,40,0], [-1,cfg.INPUT_SIZE, 21, 2])\n    left_hand = tf.where(\n            tf.math.equal(left_hand, 0.0),\n            0.0,\n            (left_hand - LEFT_HANDS_MEAN) / LEFT_HANDS_STD,\n        )\n    left_hand = tf.reshape(left_hand, [-1, cfg.INPUT_SIZE, 21*2])\n    # RIGHT HAND\n    right_hand = tf.slice(x, [0,0,61,0], [-1,cfg.INPUT_SIZE, 21, 2])\n    right_hand = tf.where(\n            tf.math.equal(right_hand, 0.0),\n            0.0,\n            (right_hand - RIGHT_HANDS_MEAN) / RIGHT_HANDS_STD,\n        )\n    right_hand = tf.reshape(right_hand, [-1, cfg.INPUT_SIZE, 21*2])\n    # POSE\n    pose = tf.slice(x, [0,0,82,0], [-1,cfg.INPUT_SIZE, 10, 2])\n    pose = tf.where(\n            tf.math.equal(pose, 0.0),\n            0.0,\n            (pose - POSE_MEAN) / POSE_STD,\n        )\n    pose = tf.reshape(pose, [-1, cfg.INPUT_SIZE, 10*2])\n    x = lips, left_hand, right_hand, pose\n    x = CustomEmbedding()(lips, left_hand, right_hand, pose, non_empty_frame_idxs)\n    # Encoder Transformer Blocks\n    x = Transformer(NUM_BLOCKS)(x, mask)\n    # Pooling\n    x = tf.reduce_sum(x * mask, axis=1) / tf.reduce_sum(mask, axis=1)\n    # Classification Layer\n    x = tf.keras.layers.Dense(cfg.NUM_CLASSES, activation=tf.keras.activations.softmax, kernel_initializer=INIT_GLOROT_UNIFORM)(x)\n    outputs = x\n    \n    # Create Tensorflow Model\n    model = tf.keras.models.Model(inputs=[frames, non_empty_frame_idxs], outputs=outputs)\n    \n    # Simple Categorical Crossentropy Loss\n    loss = tf.keras.losses.SparseCategoricalCrossentropy()\n    \n    # Adam Optimizer with weight decay\n    optimizer = tfa.optimizers.AdamW(learning_rate=1e-3, weight_decay=1e-5, clipnorm=1.0)\n    \n    lr_metric = get_lr_metric(optimizer)\n    metrics = [\"acc\",lr_metric]\n    model.compile(loss=loss, optimizer=optimizer, metrics=metrics)\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.337310Z","iopub.execute_input":"2024-02-07T04:35:34.337674Z","iopub.status.idle":"2024-02-07T04:35:34.357826Z","shell.execute_reply.started":"2024-02-07T04:35:34.337645Z","shell.execute_reply":"2024-02-07T04:35:34.356574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_one = get_model()","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:34.359308Z","iopub.execute_input":"2024-02-07T04:35:34.359725Z","iopub.status.idle":"2024-02-07T04:35:37.457535Z","shell.execute_reply.started":"2024-02-07T04:35:34.359694Z","shell.execute_reply":"2024-02-07T04:35:37.456619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_one.load_weights('/kaggle/input/model-result/artifacts/final_model_one:v1/final_model_one_weights')","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:37.459135Z","iopub.execute_input":"2024-02-07T04:35:37.459780Z","iopub.status.idle":"2024-02-07T04:35:38.998438Z","shell.execute_reply.started":"2024-02-07T04:35:37.459748Z","shell.execute_reply":"2024-02-07T04:35:38.997577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/input/asl-signs/sign_to_prediction_index_map.json', 'r') as f:\n    data = json.load(f)\n\n# Create a dictionary to map the indexes to their corresponding values\nindex_to_value = {value: str(index) for index, value in data.items()}","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:38.999499Z","iopub.execute_input":"2024-02-07T04:35:39.000013Z","iopub.status.idle":"2024-02-07T04:35:39.011244Z","shell.execute_reply.started":"2024-02-07T04:35:38.999984Z","shell.execute_reply":"2024-02-07T04:35:39.010255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.read_csv('/kaggle/input/asl-signs/train.csv')","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:39.012303Z","iopub.execute_input":"2024-02-07T04:35:39.013454Z","iopub.status.idle":"2024-02-07T04:35:39.281918Z","shell.execute_reply.started":"2024-02-07T04:35:39.013422Z","shell.execute_reply":"2024-02-07T04:35:39.280733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"train_landmark_files/26734/1000035562.parquet\n26734\n1000035562\nblow","metadata":{}},{"cell_type":"code","source":"ROWS_PER_FRAME = 543\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:54:06.008150Z","iopub.execute_input":"2024-02-07T04:54:06.008558Z","iopub.status.idle":"2024-02-07T04:54:06.014577Z","shell.execute_reply.started":"2024-02-07T04:54:06.008525Z","shell.execute_reply":"2024-02-07T04:54:06.013255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_data(file_path):\n    data = load_relevant_data_subset(os.path.join('/kaggle/input/asl-signs' , file_path))\n    data = preprocess_layer(data)\n    return data","metadata":{"execution":{"iopub.status.busy":"2024-02-07T04:35:39.283862Z","iopub.execute_input":"2024-02-07T04:35:39.284685Z","iopub.status.idle":"2024-02-07T04:35:39.290587Z","shell.execute_reply.started":"2024-02-07T04:35:39.284642Z","shell.execute_reply":"2024-02-07T04:35:39.289093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\nwith open('/kaggle/input/wlasl-processed/WLASL_v0.3.json') as f:\n    f = json.load(f)\nids = [i for i in f if i['gloss'] == 'bird']\neg = [(os.path.join('/kaggle/input/wlasl-processed/videos',i['video_id'] + '.mp4') , i['frame_start'], i['frame_end'] , i['fps']) for i in ids[0]['instances']]","metadata":{"execution":{"iopub.status.busy":"2024-02-07T05:55:42.292882Z","iopub.execute_input":"2024-02-07T05:55:42.293305Z","iopub.status.idle":"2024-02-07T05:55:43.002488Z","shell.execute_reply.started":"2024-02-07T05:55:42.293276Z","shell.execute_reply":"2024-02-07T05:55:43.001128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data, non_empty_frames_idxs = get_data('train_landmark_files/26734/1000035562.parquet')\ndata = np.array(data).reshape(1,32,92,3)\nnon_empty_frames_idxs = np.array(non_empty_frames_idxs).reshape(1,32)","metadata":{"execution":{"iopub.status.busy":"2024-02-07T05:55:47.795747Z","iopub.execute_input":"2024-02-07T05:55:47.796814Z","iopub.status.idle":"2024-02-07T05:55:47.806613Z","shell.execute_reply.started":"2024-02-07T05:55:47.796774Z","shell.execute_reply":"2024-02-07T05:55:47.805186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data = np.array(data).reshape(1,32,92,3)\ny_val_pred = model_one.predict({ 'frames': data, 'non_empty_frame_idxs': non_empty_frames_idxs })","metadata":{"execution":{"iopub.status.busy":"2024-02-07T05:55:48.062771Z","iopub.execute_input":"2024-02-07T05:55:48.063195Z","iopub.status.idle":"2024-02-07T05:55:48.167737Z","shell.execute_reply.started":"2024-02-07T05:55:48.063159Z","shell.execute_reply":"2024-02-07T05:55:48.166876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index_to_value[y_val_pred.argmax(axis = 1)[0]]","metadata":{"execution":{"iopub.status.busy":"2024-02-07T05:55:48.428146Z","iopub.execute_input":"2024-02-07T05:55:48.428872Z","iopub.status.idle":"2024-02-07T05:55:48.434998Z","shell.execute_reply.started":"2024-02-07T05:55:48.428835Z","shell.execute_reply":"2024-02-07T05:55:48.434003Z"},"trusted":true},"execution_count":null,"outputs":[]}]}