{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# importing Libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sn\n\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split, GroupShuffleSplit \n\nimport glob\nimport sys\nimport os\nimport math\nimport gc\nimport sys\nimport sklearn\nimport scipy\n\nprint(f'Tensorflow V{tf.__version__}')\nprint(f'Keras V{tf.keras.__version__}')\nprint(f'Python V{sys.version}')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:01.541795Z","iopub.execute_input":"2023-04-26T19:34:01.542250Z","iopub.status.idle":"2023-04-26T19:34:10.953434Z","shell.execute_reply.started":"2023-04-26T19:34:01.542205Z","shell.execute_reply":"2023-04-26T19:34:10.950993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Matplot lib Parameter config","metadata":{}},{"cell_type":"code","source":"# MatplotLib Global Settings\nmpl.rcParams.update(mpl.rcParamsDefault)\nmpl.rcParams['xtick.labelsize'] = 16\nmpl.rcParams['ytick.labelsize'] = 16\nmpl.rcParams['axes.labelsize'] = 18\nmpl.rcParams['axes.titlesize'] = 24","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:10.956100Z","iopub.execute_input":"2023-04-26T19:34:10.956923Z","iopub.status.idle":"2023-04-26T19:34:10.966066Z","shell.execute_reply.started":"2023-04-26T19:34:10.956882Z","shell.execute_reply":"2023-04-26T19:34:10.963170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training parameter config","metadata":{}},{"cell_type":"code","source":"# If True, processing data from scratch\n# If False, loads preprocessed data\nPREPROCESS_DATA = False\nTRAIN_MODEL = True\n# True: use 10% of participants as validation set\n# False: use all data for training -> gives better LB result\nUSE_VAL = False\n\nN_ROWS = 543\nN_DIMS = 3\nDIM_NAMES = ['x', 'y', 'z']\nSEED = 42\nNUM_CLASSES = 250\nIS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\nVERBOSE = 1 if IS_INTERACTIVE else 2\n\nINPUT_SIZE = 64\n\nBATCH_ALL_SIGNS_N = 4\nBATCH_SIZE = 256\nN_EPOCHS = 100\nLR_MAX = 1e-3\nN_WARMUP_EPOCHS = 0\nWD_RATIO = 0.05\nMASK_VAL = 4237","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:10.967240Z","iopub.execute_input":"2023-04-26T19:34:10.967519Z","iopub.status.idle":"2023-04-26T19:34:10.984262Z","shell.execute_reply.started":"2023-04-26T19:34:10.967492Z","shell.execute_reply":"2023-04-26T19:34:10.982997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utils","metadata":{}},{"cell_type":"code","source":"# Prints Shape and Dtype For List Of Variables\ndef print_shape_dtype(l, names):\n    for e, n in zip(l, names):\n        print(f'{n} shape: {e.shape}, dtype: {e.dtype}')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:10.986125Z","iopub.execute_input":"2023-04-26T19:34:10.986541Z","iopub.status.idle":"2023-04-26T19:34:10.996084Z","shell.execute_reply.started":"2023-04-26T19:34:10.986487Z","shell.execute_reply":"2023-04-26T19:34:10.994910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train\n","metadata":{}},{"cell_type":"code","source":"# Read Training Data\nif IS_INTERACTIVE or not PREPROCESS_DATA:\n    train = pd.read_csv('/kaggle/input/asl-signs/train.csv').sample(int(5e3), random_state=SEED)\nelse:\n    train = pd.read_csv('/kaggle/input/asl-signs/train.csv')\n\nN_SAMPLES = len(train)\nprint(f'N_SAMPLES: {N_SAMPLES}')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:11.000376Z","iopub.execute_input":"2023-04-26T19:34:11.000722Z","iopub.status.idle":"2023-04-26T19:34:11.211189Z","shell.execute_reply.started":"2023-04-26T19:34:11.000651Z","shell.execute_reply":"2023-04-26T19:34:11.209887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read file path \n\nThis code defines a function called get_file_path() that takes one argument path and returns a full file path. This function is used to convert a file path to a file path in the Kaggle dataset. The code applies the get_file_path() function to each element in the Pandas Series object named train['path'] using Pandas' apply() function. The code then stores the result in a new column named train['file_path']。","metadata":{}},{"cell_type":"markdown","source":"# Add file Path","metadata":{}},{"cell_type":"code","source":"# Get complete file path to file\ndef get_file_path(path):\n    return f'/kaggle/input/asl-signs/{path}'\n\ntrain['file_path'] = train['path'].apply(get_file_path)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:11.212906Z","iopub.execute_input":"2023-04-26T19:34:11.213286Z","iopub.status.idle":"2023-04-26T19:34:11.227343Z","shell.execute_reply.started":"2023-04-26T19:34:11.213245Z","shell.execute_reply":"2023-04-26T19:34:11.226210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ordinally Encode signs","metadata":{}},{"cell_type":"code","source":"# Add ordinally Encoded Sign (assign number to each sign name)\ntrain['sign_ord'] = train['sign'].astype('category').cat.codes\n\n# Dictionaries to translate sign <-> ordinal encoded sign\nSIGN2ORD = train[['sign', 'sign_ord']].set_index('sign').squeeze().to_dict()\nORD2SIGN = train[['sign_ord', 'sign']].set_index('sign_ord').squeeze().to_dict()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:11.229126Z","iopub.execute_input":"2023-04-26T19:34:11.229643Z","iopub.status.idle":"2023-04-26T19:34:11.253377Z","shell.execute_reply.started":"2023-04-26T19:34:11.229602Z","shell.execute_reply":"2023-04-26T19:34:11.252434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train.head(30))\ndisplay(train.info())","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:11.256338Z","iopub.execute_input":"2023-04-26T19:34:11.257462Z","iopub.status.idle":"2023-04-26T19:34:11.294797Z","shell.execute_reply.started":"2023-04-26T19:34:11.257415Z","shell.execute_reply":"2023-04-26T19:34:11.293425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Satistical video data\n\nThis code is used to calculate the statistics of the number of different frames, the number of missing frames and the maximum number of frames for each video in the video dataset. Among them, N is the value set according to the condition, and N_UNIQUE_FRAMES, N_MISSING_FRAMES, and MAX_FRAME are arrays storing the number of different frames, the number of missing frames, and the maximum number of frames, respectively. In the loop, the code iterates through each video in the dataset, reading the video data and counting the number of distinct frames, missing frames, and maximum frames. Finally, the code displays these statistics and draws a corresponding histogram.","metadata":{}},{"cell_type":"markdown","source":"# Video Stats","metadata":{}},{"cell_type":"code","source":"N = int(1e3) if (IS_INTERACTIVE or not PREPROCESS_DATA) else int(10e3)\nN_UNIQUE_FRAMES = np.zeros(N, dtype=np.uint16)\nN_MISSING_FRAMES = np.zeros(N, dtype=np.uint16)\nMAX_FRAME = np.zeros(N, dtype=np.uint16)\n\nPERCENTILES = [0.01, 0.05, 0.25, 0.50, 0.75, 0.95, 0.99, 0.999]\n\nfor idx, file_path in enumerate(tqdm(train['file_path'].sample(N, random_state=SEED))):\n    df = pd.read_parquet(file_path)\n    N_UNIQUE_FRAMES[idx] = df['frame'].nunique()\n    N_MISSING_FRAMES[idx] = (df['frame'].max() - df['frame'].min()) - df['frame'].nunique() + 1\n    MAX_FRAME[idx] = df['frame'].max()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:11.296712Z","iopub.execute_input":"2023-04-26T19:34:11.297493Z","iopub.status.idle":"2023-04-26T19:34:33.773862Z","shell.execute_reply.started":"2023-04-26T19:34:11.297430Z","shell.execute_reply":"2023-04-26T19:34:33.772610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of unique frames in each video\ndisplay(pd.Series(N_UNIQUE_FRAMES).describe(percentiles=PERCENTILES).to_frame('N_UNIQUE_FRAMES'))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Unique Frames', size=24)\npd.Series(N_UNIQUE_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nxlim = math.ceil(plt.xlim()[1])\nplt.xlim(0, xlim)\nplt.xticks(np.arange(0, xlim+25, 25))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:33.775703Z","iopub.execute_input":"2023-04-26T19:34:33.776413Z","iopub.status.idle":"2023-04-26T19:34:34.325999Z","shell.execute_reply.started":"2023-04-26T19:34:33.776365Z","shell.execute_reply":"2023-04-26T19:34:34.324998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of missing frames, consecutive frames with missing intermediate frame, i.e. 1,2,4,5 -> 3 is missing\ndisplay(pd.Series(N_MISSING_FRAMES).describe(percentiles=PERCENTILES).to_frame('N_MISSING_FRAMES'))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Missing Frames', size=24)\npd.Series(N_MISSING_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nplt.xlim(0, math.ceil(plt.xlim()[1]))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:34.327510Z","iopub.execute_input":"2023-04-26T19:34:34.328630Z","iopub.status.idle":"2023-04-26T19:34:34.785552Z","shell.execute_reply.started":"2023-04-26T19:34:34.328580Z","shell.execute_reply":"2023-04-26T19:34:34.784500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Maximum frame number\ndisplay(pd.Series(MAX_FRAME).describe(percentiles=PERCENTILES).to_frame('MAX_FRAME'))\n\nplt.figure(figsize=(15,8))\nplt.title('Maximum Frames Index', size=24)\npd.Series(MAX_FRAME).plot(kind='hist', bins=128)\nplt.grid()\nplt.xlim(0, math.ceil(plt.xlim()[1]))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:34.787099Z","iopub.execute_input":"2023-04-26T19:34:34.790557Z","iopub.status.idle":"2023-04-26T19:34:35.297392Z","shell.execute_reply.started":"2023-04-26T19:34:34.790505Z","shell.execute_reply":"2023-04-26T19:34:35.296232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Landmark Indices key point index\nIn machine learning, Landmark Indices usually refers to the index of face key points. Face key points are some specific points on the face, such as eyes, nose, mouth, etc., which can be used for tasks such as face recognition and expression recognition. In machine learning, we can use these key points to train models to achieve tasks such as face recognition. ¹","metadata":{}},{"cell_type":"markdown","source":"# Landmark Indices","metadata":{}},{"cell_type":"code","source":"USE_TYPES = ['left_hand', 'pose', 'right_hand']\nSTART_IDX = 468\nLIPS_IDXS0 = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\n# Landmark indices in original data\nLEFT_HAND_IDXS0 = np.arange(468,489)\nRIGHT_HAND_IDXS0 = np.arange(522,543)\nLEFT_POSE_IDXS0 = np.array([502, 504, 506, 508, 510])\nRIGHT_POSE_IDXS0 = np.array([503, 505, 507, 509, 511])\nLANDMARK_IDXS_LEFT_DOMINANT0 = np.concatenate((LIPS_IDXS0, LEFT_HAND_IDXS0, LEFT_POSE_IDXS0))\nLANDMARK_IDXS_RIGHT_DOMINANT0 = np.concatenate((LIPS_IDXS0, RIGHT_HAND_IDXS0, RIGHT_POSE_IDXS0))\nHAND_IDXS0 = np.concatenate((LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0), axis=0)\nN_COLS = LANDMARK_IDXS_LEFT_DOMINANT0.size\n# Landmark indices in processed data\nLIPS_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LIPS_IDXS0)).squeeze()\nLEFT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_HAND_IDXS0)).squeeze()\nRIGHT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, RIGHT_HAND_IDXS0)).squeeze()\nHAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, HAND_IDXS0)).squeeze()\nPOSE_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_POSE_IDXS0)).squeeze()\n\nprint(f'# HAND_IDXS: {len(HAND_IDXS)}, N_COLS: {N_COLS}')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:35.299444Z","iopub.execute_input":"2023-04-26T19:34:35.299936Z","iopub.status.idle":"2023-04-26T19:34:35.315040Z","shell.execute_reply.started":"2023-04-26T19:34:35.299895Z","shell.execute_reply":"2023-04-26T19:34:35.313718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LIPS_START = 0\nLEFT_HAND_START = LIPS_IDXS.size\nRIGHT_HAND_START = LEFT_HAND_START + LEFT_HAND_IDXS.size\nPOSE_START = RIGHT_HAND_START + RIGHT_HAND_IDXS.size\n\nprint(f'LIPS_START: {LIPS_START}, LEFT_HAND_START: {LEFT_HAND_START}, RIGHT_HAND_START: {RIGHT_HAND_START}, POSE_START: {POSE_START}')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:35.321836Z","iopub.execute_input":"2023-04-26T19:34:35.322103Z","iopub.status.idle":"2023-04-26T19:34:35.328224Z","shell.execute_reply.started":"2023-04-26T19:34:35.322076Z","shell.execute_reply":"2023-04-26T19:34:35.327090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process Data Tensorflow","metadata":{}},{"cell_type":"code","source":"# Source: https://www.kaggle.com/competitions/asl-signs/overview/evaluation\nROWS_PER_FRAME = 543  # number of landmarks per frame\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:35.330230Z","iopub.execute_input":"2023-04-26T19:34:35.330971Z","iopub.status.idle":"2023-04-26T19:34:35.338376Z","shell.execute_reply.started":"2023-04-26T19:34:35.330927Z","shell.execute_reply":"2023-04-26T19:34:35.337092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\nThis code defines a constant called ROWS_PER_FRAME with a value of 543, representing the number of landmarks per frame. The function load_relevant_data_subset(pq_path) reads the parquet file under the specified path, extracts the three columns of data x, y, and z in the data set, divides the data set according to the number of landmarks in each frame, and returns a three-dimensional array. The first dimension represents the number of frames, the second dimension represents the number of landmarks per frame, and the third dimension represents the three coordinate axes of x, y, and z. The return value type of the function is numpy.ndarray, and the data type is np.float32.","metadata":{}},{"cell_type":"markdown","source":"# Coustomize a Data processing layer using Tf\n\nhe PreprocessLayer class is a custom layer inherited from TensorFlow's tf.keras.layers.Layer class for processing data in TensorFlow Lite models. This custom layer contains functions for processing input data\n\nadd Codeadd Markdown\nFirst, the layer creates a constant called normalization_correction in the init method. The constant is a matrix with the number of rows equal to the number of marker points of a particular type in the data, and the number of columns is 3 (that is, x, y, and z coordinates). This matrix is ​​used to correct the shooting direction of the camera, adjusting the left hand to the right hand and the right hand to the left hand.\n\nThis layer also defines a method named pad_edge for padding a given tensor with a certain number of repeated elements to the left or right. Next, the layer decorates a call method with the @tf.function decorator to process the input data.\n\nThis method first calculates the number of frames of the input data (N_FRAMES0), and then finds the landmark points of the dominant hand in the data by calculating the sum of the coordinates of the left and right hands in the data. Next, the method counts the number of non-NaN values ​​in the dominant hands for each frame to determine which frames to keep. It then uses these indices to gather landmark data from the input data.\n\nThe method next converts the data type of the frame index from integer to float, and then normalizes it to start with 0. Next, it again counted the number of frames (N_FRAMES) of filtered data, and then collected certain types of landmark data from these data. If the number of frames of data is smaller than the specified input size (INPUT_SIZE), pad with -1, expand the number of frames of data to the specified input size, and replace NaN values ​​with 0. If the number of frames of data is larger than the specified input size, it is reduced to the specified input size using duplicates and any missing data is filled.\n\nFinally, the method returns the processed data and the corresponding frame index.\n\nadd Codeadd Markdown\nThis code is a TensorFlow Keras custom layer named PreprocessLayer. It is used to preprocess the input data, including processing, filling and normalizing hand keypoint data.\n\nThe main functions of this layer include:\n\nInitialization operation: In the init method, initialize by calling the init method of the parent class tf.keras.layers.Layer, and define a constant tensor named normalization_correction, and store its transposition in self.normalization_correction .\n\npad_edge method: used to pad data at the edge of the input data, according to the specified padding direction ('LEFT' or 'RIGHT') and the number of repetitions of padding.\n\ncall method: By using the @tf.function decorator, a calculation graph (Graph) operation is defined to process the input data. Specific steps are as follows:\n\na. Get the first dimension (number of frames) of the input data and store in the N_FRAMES0 variable.\n\nb. Determine whether the left hand or the right hand is the dominant hand, and according to the result of the dominant hand, calculate the sum of the non-NaN (non-empty) values ​​of the key points of the hand in each frame, and store it in left_hand_sum and right_hand_sum.\n\nc. Based on the result of the dominant hand, calculate the sum of the non-NaN (non-null) values ​​of the keypoints of the dominant hand's hand in each frame, and store it in frames_hands_non_nan_sum.\n\nd. According to the result in frames_hands_non_nan_sum, find the index of the non-empty frame and store it in non_empty_frames_idxs.\n\ne. Filter the input data according to non_empty_frames_idxs, and perform a series of normalization and filling operations, and finally return the processed data and the filled frame index.\n\nIn general, the PreprocessLayer custom layer is mainly used to preprocess the input data, including the processing, filling and normalization of hand key point data, so as to meet the input requirements of subsequent models.","metadata":{}},{"cell_type":"code","source":"\"\"\"\n    Tensorflow layer to process data in TFLite\n    Data needs to be processed in the model itself, so we can not use Python\n\"\"\" \nclass PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__()\n        normalisation_correction = tf.constant([\n                    # Add 0.50 to left hand (original right hand) and substract 0.50 of right hand (original left hand)\n                    [0] * len(LIPS_IDXS) + [0.50] * len(LEFT_HAND_IDXS) + [0.50] * len(POSE_IDXS),\n                    # Y coordinates stay intact\n                    [0] * len(LANDMARK_IDXS_LEFT_DOMINANT0),\n                    # Z coordinates stay intact\n                    [0] * len(LANDMARK_IDXS_LEFT_DOMINANT0),\n                ],\n                dtype=tf.float32,\n            )\n        self.normalisation_correction = tf.transpose(normalisation_correction, [1,0])\n        \n    def pad_edge(self, t, repeats, side):\n        if side == 'LEFT':\n            return tf.concat((tf.repeat(t[:1], repeats=repeats, axis=0), t), axis=0)\n        elif side == 'RIGHT':\n            return tf.concat((t, tf.repeat(t[-1:], repeats=repeats, axis=0)), axis=0)\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,N_ROWS,N_DIMS], dtype=tf.float32),),\n    )\n    def call(self, data0):\n        # Number of Frames in Video\n        N_FRAMES0 = tf.shape(data0)[0]\n        \n        # Find dominant hand by comparing summed absolute coordinates\n        left_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0, 1))\n        right_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0, 1))\n        left_dominant = left_hand_sum >= right_hand_sum\n        \n        # Count non NaN Hand values in each frame for the dominant hand\n        if left_dominant:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0, 1),\n                    axis=[1, 2],\n                )\n        else:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0, 1),\n                    axis=[1, 2],\n                )\n        \n        # Find frames indices with coordinates of dominant hand\n        non_empty_frames_idxs = tf.where(frames_hands_non_nan_sum > 0)\n        non_empty_frames_idxs = tf.squeeze(non_empty_frames_idxs, axis=1)\n        # Filter frames\n        data = tf.gather(data0, non_empty_frames_idxs, axis=0)\n        \n        # Cast Indices in float32 to be compatible with Tensorflow Lite\n        non_empty_frames_idxs = tf.cast(non_empty_frames_idxs, tf.float32)\n        # Normalize to start with 0\n        non_empty_frames_idxs -= tf.reduce_min(non_empty_frames_idxs)\n        \n        # Number of Frames in Filtered Video\n        N_FRAMES = tf.shape(data)[0]\n        \n        # Gather Relevant Landmark Columns\n        if left_dominant:\n            data = tf.gather(data, LANDMARK_IDXS_LEFT_DOMINANT0, axis=1)\n        else:\n            data = tf.gather(data, LANDMARK_IDXS_RIGHT_DOMINANT0, axis=1)\n            data = (\n                    self.normalisation_correction + (\n                        (data - self.normalisation_correction) * tf.where(self.normalisation_correction != 0, -1.0, 1.0))\n                )\n        \n        # Video fits in INPUT_SIZE\n        if N_FRAMES < INPUT_SIZE:\n            # Pad With -1 to indicate padding\n            non_empty_frames_idxs = tf.pad(non_empty_frames_idxs, [[0, INPUT_SIZE-N_FRAMES]], constant_values=-1)\n            # Pad Data With Zeros\n            data = tf.pad(data, [[0, INPUT_SIZE-N_FRAMES], [0,0], [0,0]], constant_values=0)\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            return data, non_empty_frames_idxs\n        # Video needs to be downsampled to INPUT_SIZE\n        else:\n            # Repeat\n            if N_FRAMES < INPUT_SIZE**2:\n                repeats = tf.math.floordiv(INPUT_SIZE * INPUT_SIZE, N_FRAMES0)\n                data = tf.repeat(data, repeats=repeats, axis=0)\n                non_empty_frames_idxs = tf.repeat(non_empty_frames_idxs, repeats=repeats, axis=0)\n\n            # Pad To Multiple Of Input Size\n            pool_size = tf.math.floordiv(len(data), INPUT_SIZE)\n            if tf.math.mod(len(data), INPUT_SIZE) > 0:\n                pool_size += 1\n\n            if pool_size == 1:\n                pad_size = (pool_size * INPUT_SIZE) - len(data)\n            else:\n                pad_size = (pool_size * INPUT_SIZE) % len(data)\n\n            # Pad Start/End with Start/End value\n            pad_left = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(INPUT_SIZE, 2)\n            pad_right = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(INPUT_SIZE, 2)\n            if tf.math.mod(pad_size, 2) > 0:\n                pad_right += 1\n\n            # Pad By Concatenating Left/Right Edge Values\n            data = self.pad_edge(data, pad_left, 'LEFT')\n            data = self.pad_edge(data, pad_right, 'RIGHT')\n\n            # Pad Non Empty Frame Indices\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_left, 'LEFT')\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_right, 'RIGHT')\n\n            # Reshape to Mean Pool\n            data = tf.reshape(data, [INPUT_SIZE, -1, N_COLS, N_DIMS])\n            non_empty_frames_idxs = tf.reshape(non_empty_frames_idxs, [INPUT_SIZE, -1])\n\n            # Mean Pool\n            data = tf.experimental.numpy.nanmean(data, axis=1)\n            non_empty_frames_idxs = tf.experimental.numpy.nanmean(non_empty_frames_idxs, axis=1)\n\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            \n            return data, non_empty_frames_idxs\n    \npreprocess_layer = PreprocessLayer()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:35.339929Z","iopub.execute_input":"2023-04-26T19:34:35.340744Z","iopub.status.idle":"2023-04-26T19:34:38.949037Z","shell.execute_reply.started":"2023-04-26T19:34:35.340704Z","shell.execute_reply":"2023-04-26T19:34:38.948010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Interpolate NaN Values\nInterpolate NaN Values ​​refers to the use of interpolation to fill the NaN values ​​​​in the data. NaN is a class of values ​​of numeric data types in computer science that represent undefined or unrepresentable values. NaN is the abbreviation of Not a Number, understood as not a value. In computers, NaN is usually used to represent invalid or undefined operation results, such as 0/0, ∞-∞, etc. 1.","metadata":{}},{"cell_type":"code","source":"\"\"\"\n    face: 0:468\n    left_hand: 468:489\n    pose: 489:522\n    right_hand: 522:544\n        \n\"\"\"\ndef get_data(file_path):\n    # Load Raw Data\n    data = load_relevant_data_subset(file_path)\n    # Process Data Using Tensorflow\n    data = preprocess_layer(data)\n    \n    return data","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:38.950594Z","iopub.execute_input":"2023-04-26T19:34:38.950979Z","iopub.status.idle":"2023-04-26T19:34:38.958325Z","shell.execute_reply.started":"2023-04-26T19:34:38.950940Z","shell.execute_reply":"2023-04-26T19:34:38.957086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Dataset","metadata":{}},{"cell_type":"code","source":"# Get the full dataset\ndef preprocess_data():\n    # Create arrays to save data\n    X = np.zeros([N_SAMPLES, INPUT_SIZE, N_COLS, N_DIMS], dtype=np.float32)\n    y = np.zeros([N_SAMPLES], dtype=np.int32)\n    NON_EMPTY_FRAME_IDXS = np.full([N_SAMPLES, INPUT_SIZE], -1, dtype=np.float32)\n\n    # Fill X/y\n    for row_idx, (file_path, sign_ord) in enumerate(tqdm(train[['file_path', 'sign_ord']].values)):\n        # Log message every 5000 samples\n        if row_idx % 5000 == 0:\n            print(f'Generated {row_idx}/{N_SAMPLES}')\n\n        data, non_empty_frame_idxs = get_data(file_path)\n        X[row_idx] = data\n        y[row_idx] = sign_ord\n        NON_EMPTY_FRAME_IDXS[row_idx] = non_empty_frame_idxs\n        # Sanity check, data should not contain NaN values\n        if np.isnan(data).sum() > 0:\n            print(row_idx)\n            return data\n\n    # Save X/y\n    np.save('X.npy', X)\n    np.save('y.npy', y)\n    np.save('NON_EMPTY_FRAME_IDXS.npy', NON_EMPTY_FRAME_IDXS)\n    \n    # Save Validation\n    splitter = GroupShuffleSplit(test_size=0.10, n_splits=2, random_state=SEED)\n    PARTICIPANT_IDS = train['participant_id'].values\n    train_idxs, val_idxs = next(splitter.split(X, y, groups=PARTICIPANT_IDS))\n\n    # Save Train\n    X_train = X[train_idxs]\n    NON_EMPTY_FRAME_IDXS_TRAIN = NON_EMPTY_FRAME_IDXS[train_idxs]\n    y_train = y[train_idxs]\n    np.save('X_train.npy', X_train)\n    np.save('y_train.npy', y_train)\n    np.save('NON_EMPTY_FRAME_IDXS_TRAIN.npy', NON_EMPTY_FRAME_IDXS_TRAIN)\n    # Save Validation\n    X_val = X[val_idxs]\n    NON_EMPTY_FRAME_IDXS_VAL = NON_EMPTY_FRAME_IDXS[val_idxs]\n    y_val = y[val_idxs]\n    np.save('X_val.npy', X_val)\n    np.save('y_val.npy', y_val)\n    np.save('NON_EMPTY_FRAME_IDXS_VAL.npy', NON_EMPTY_FRAME_IDXS_VAL)\n    # Split Statistics\n    print(f'Patient ID Intersection Train/Val: {set(PARTICIPANT_IDS[train_idxs]).intersection(PARTICIPANT_IDS[val_idxs])}')\n    print(f'X_train shape: {X_train.shape}, X_val shape: {X_val.shape}')\n    print(f'y_train shape: {y_train.shape}, y_val shape: {y_val.shape}')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:38.960402Z","iopub.execute_input":"2023-04-26T19:34:38.960802Z","iopub.status.idle":"2023-04-26T19:34:38.974306Z","shell.execute_reply.started":"2023-04-26T19:34:38.960762Z","shell.execute_reply":"2023-04-26T19:34:38.973200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess All Data From Scratch\nif PREPROCESS_DATA:\n    preprocess_data()\n    ROOT_DIR = '.'\nelse:\n    ROOT_DIR = '/kaggle/input/gislr-dataset-public'\n    \n# Load Data\nif USE_VAL:\n    # Load Train\n    X_train = np.load(f'{ROOT_DIR}/X_train.npy')\n    y_train = np.load(f'{ROOT_DIR}/y_train.npy')\n    NON_EMPTY_FRAME_IDXS_TRAIN = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS_TRAIN.npy')\n    # Load Val\n    X_val = np.load(f'{ROOT_DIR}/X_val.npy')\n    y_val = np.load(f'{ROOT_DIR}/y_val.npy')\n    NON_EMPTY_FRAME_IDXS_VAL = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS_VAL.npy')\n    # Define validation Data\n    validation_data = ({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, y_val)\nelse:\n    X_train = np.load(f'{ROOT_DIR}/X.npy')\n    y_train = np.load(f'{ROOT_DIR}/y.npy')\n    NON_EMPTY_FRAME_IDXS_TRAIN = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS.npy')\n    validation_data = None\n\n# Train \nprint_shape_dtype([X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN], ['X_train', 'y_train', 'NON_EMPTY_FRAME_IDXS_TRAIN'])\n# Val\nif USE_VAL:\n    print_shape_dtype([X_val, y_val, NON_EMPTY_FRAME_IDXS_VAL], ['X_val', 'y_val', 'NON_EMPTY_FRAME_IDXS_VAL'])\n# Sanity Check\nprint(f'# NaN Values X_train: {np.isnan(X_train).sum()}')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:34:38.975832Z","iopub.execute_input":"2023-04-26T19:34:38.976255Z","iopub.status.idle":"2023-04-26T19:35:15.753524Z","shell.execute_reply.started":"2023-04-26T19:34:38.976220Z","shell.execute_reply":"2023-04-26T19:35:15.752234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Class Count\ndisplay(pd.Series(y_train).value_counts().to_frame('Class Count').iloc[[0,1,2,3,4, -5,-4,-3,-2,-1]])","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:35:15.755194Z","iopub.execute_input":"2023-04-26T19:35:15.758377Z","iopub.status.idle":"2023-04-26T19:35:15.772626Z","shell.execute_reply.started":"2023-04-26T19:35:15.758343Z","shell.execute_reply":"2023-04-26T19:35:15.771563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Number of Frames","metadata":{}},{"cell_type":"code","source":"# Vast majority of samples fits has less than 32 non empty frames\nN_EMPTY_FRAMES = (NON_EMPTY_FRAME_IDXS_TRAIN != -1).sum(axis=1) \nN_EMPTY_FRAMES_WATERFALL = []\nfor n in tqdm(range(1,INPUT_SIZE+1)):\n    N_EMPTY_FRAMES_WATERFALL.append(sum(N_EMPTY_FRAMES >= n) / len(NON_EMPTY_FRAME_IDXS_TRAIN) * 100)\n\nplt.figure(figsize=(18,10))\nplt.title('Waterfall Plot For Number Of Non Empty Frames')\npd.Series(N_EMPTY_FRAMES_WATERFALL).plot(kind='bar')\nplt.grid(axis='y')\nplt.xticks(np.arange(INPUT_SIZE), np.arange(1, INPUT_SIZE+1))\nplt.xlabel('Number of Non Empty Frames', size=16)\nplt.yticks(np.arange(0, 100+10, 10))\nplt.ylim(0, 100)\nplt.ylabel('Percentage of Samples With At Least N Non Empty Frames', size=16)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:35:15.773977Z","iopub.execute_input":"2023-04-26T19:35:15.775214Z","iopub.status.idle":"2023-04-26T19:35:29.760561Z","shell.execute_reply.started":"2023-04-26T19:35:15.775162Z","shell.execute_reply":"2023-04-26T19:35:29.759481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Percentages of Frames Filled","metadata":{}},{"cell_type":"code","source":"# Percentage of frames filled, this is the maximum fill percentage of each landmark\nP_DATA_FILLED = (NON_EMPTY_FRAME_IDXS_TRAIN != -1).sum() / NON_EMPTY_FRAME_IDXS_TRAIN.size * 100\nprint(f'P_DATA_FILLED: {P_DATA_FILLED:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:35:29.761969Z","iopub.execute_input":"2023-04-26T19:35:29.762366Z","iopub.status.idle":"2023-04-26T19:35:29.778105Z","shell.execute_reply.started":"2023-04-26T19:35:29.762327Z","shell.execute_reply":"2023-04-26T19:35:29.777245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Statistics -Lips","metadata":{}},{"cell_type":"code","source":"# Percentage of Lips Measurements\nP_LEFT_LIPS_MEASUREMENTS = (X_train[:,:,LIPS_IDXS] != 0).sum() / X_train[:,:,LIPS_IDXS].size / P_DATA_FILLED * 1e4\nprint(f'P_LEFT_LIPS_MEASUREMENTS: {P_LEFT_LIPS_MEASUREMENTS:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:35:29.779337Z","iopub.execute_input":"2023-04-26T19:35:29.780116Z","iopub.status.idle":"2023-04-26T19:35:47.570665Z","shell.execute_reply.started":"2023-04-26T19:35:29.780086Z","shell.execute_reply":"2023-04-26T19:35:47.569458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"What this code does is calculate the mean and standard deviation of the lips. It uses numpy and matplotlib libraries. Among them, the np.transpose function converts the dimension of the X_train array from (number of samples, number of time steps, number of features) to (number of features, number of time steps, number of samples), and then the reshape function converts it into (number of features, number of time steps number * number of samples) shape. Finally, for each feature and each sample, compute the mean and standard deviation of the nonzero elements and store the results in the LIPS_MEAN_X, LIPS_MEAN_Y, LIPS_STD_X, and LIPS_STD_Y arrays. These arrays are finally combined into LIPS_MEAN and LIPS_STD arrays and returned as the output of the function.","metadata":{}},{"cell_type":"code","source":"def get_lips_mean_std():\n    # LIPS\n    LIPS_MEAN_X = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_MEAN_Y = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_STD_X = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_STD_Y = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n\n    fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,LIPS_IDXS], [2,3,0,1]).reshape([LIPS_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                LIPS_MEAN_X[col] = v.mean()\n                LIPS_STD_X[col] = v.std()\n            if dim == 1: # Y\n                LIPS_MEAN_Y[col] = v.mean()\n                LIPS_STD_Y[col] = v.std()\n\n            axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    for ax, dim_name in zip(axes, DIM_NAMES):\n        ax.set_title(f'Lips {dim_name.upper()} Dimension', size=24)\n        ax.tick_params(axis='x', labelsize=8)\n        ax.grid(axis='y')\n\n    plt.subplots_adjust(hspace=0.50)\n    plt.show()\n\n    LIPS_MEAN = np.array([LIPS_MEAN_X, LIPS_MEAN_Y]).T\n    LIPS_STD = np.array([LIPS_STD_X, LIPS_STD_Y]).T\n    \n    return LIPS_MEAN, LIPS_STD\n\nLIPS_MEAN, LIPS_STD = get_lips_mean_std()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:35:47.572700Z","iopub.execute_input":"2023-04-26T19:35:47.573083Z","iopub.status.idle":"2023-04-26T19:36:07.831325Z","shell.execute_reply.started":"2023-04-26T19:35:47.573043Z","shell.execute_reply":"2023-04-26T19:36:07.830348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Statistics - Hands","metadata":{}},{"cell_type":"code","source":"# Verify Normalised to Left Hand Dominant\nP_LEFT_HAND_MEASUREMENTS = (X_train[:,:,LEFT_HAND_IDXS] != 0).sum() / X_train[:,:,LEFT_HAND_IDXS].size / P_DATA_FILLED * 1e4\n# P_RIGHT_HAND_MEASUREMENTS = (X_train[:,:,RIGHT_HAND_IDXS] != 0).sum() / X_train[:,:,RIGHT_HAND_IDXS].size / P_DATA_FILLED * 1e4\nprint(f'P_LEFT_HAND_MEASUREMENTS: {P_LEFT_HAND_MEASUREMENTS:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:07.834387Z","iopub.execute_input":"2023-04-26T19:36:07.834856Z","iopub.status.idle":"2023-04-26T19:36:17.753956Z","shell.execute_reply.started":"2023-04-26T19:36:07.834818Z","shell.execute_reply":"2023-04-26T19:36:17.751909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_left_right_hand_mean_std():\n    # LEFT HAND\n    LEFT_HANDS_MEAN_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_MEAN_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_STD_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_STD_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n\n    fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,LEFT_HAND_IDXS], [2,3,0,1]).reshape([LEFT_HAND_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                LEFT_HANDS_MEAN_X[col] = v.mean()\n                LEFT_HANDS_STD_X[col] = v.std()\n            if dim == 1: # Y\n                LEFT_HANDS_MEAN_Y[col] = v.mean()\n                LEFT_HANDS_STD_Y[col] = v.std()\n            # Plot\n            axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    for ax, dim_name in zip(axes, DIM_NAMES):\n        ax.set_title(f'Hands {dim_name.upper()} Dimension', size=24)\n        ax.tick_params(axis='x', labelsize=8)\n        ax.grid(axis='y')\n\n    plt.subplots_adjust(hspace=0.50)\n    plt.show()\n\n    LEFT_HANDS_MEAN = np.array([LEFT_HANDS_MEAN_X, LEFT_HANDS_MEAN_Y]).T\n    LEFT_HANDS_STD = np.array([LEFT_HANDS_STD_X, LEFT_HANDS_STD_Y]).T\n    \n    return LEFT_HANDS_MEAN, LEFT_HANDS_STD\n\nLEFT_HANDS_MEAN, LEFT_HANDS_STD = get_left_right_hand_mean_std()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:17.755297Z","iopub.execute_input":"2023-04-26T19:36:17.755684Z","iopub.status.idle":"2023-04-26T19:36:29.795886Z","shell.execute_reply.started":"2023-04-26T19:36:17.755645Z","shell.execute_reply":"2023-04-26T19:36:29.792600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Statistics - Pose\n","metadata":{}},{"cell_type":"code","source":"# Percentage of Lips Measurements\nP_POSE_MEASUREMENTS = (X_train[:,:,POSE_IDXS] != 0).sum() / X_train[:,:,POSE_IDXS].size / P_DATA_FILLED * 1e4\nprint(f'P_POSE_MEASUREMENTS: {P_POSE_MEASUREMENTS:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:29.797336Z","iopub.execute_input":"2023-04-26T19:36:29.798809Z","iopub.status.idle":"2023-04-26T19:36:31.798546Z","shell.execute_reply.started":"2023-04-26T19:36:29.798758Z","shell.execute_reply":"2023-04-26T19:36:31.797256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_pose_mean_std():\n    # POSE\n    POSE_MEAN_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_MEAN_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_STD_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_STD_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\n\n    fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,POSE_IDXS], [2,3,0,1]).reshape([POSE_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                POSE_MEAN_X[col] = v.mean()\n                POSE_STD_X[col] = v.std()\n            if dim == 1: # Y\n                POSE_MEAN_Y[col] = v.mean()\n                POSE_STD_Y[col] = v.std()\n\n            axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    for ax, dim_name in zip(axes, DIM_NAMES):\n        ax.set_title(f'Pose {dim_name.upper()} Dimension', size=24)\n        ax.tick_params(axis='x', labelsize=8)\n        ax.grid(axis='y')\n\n    plt.subplots_adjust(hspace=0.50)\n    plt.show()\n\n    POSE_MEAN = np.array([POSE_MEAN_X, POSE_MEAN_Y]).T\n    POSE_STD = np.array([POSE_STD_X, POSE_STD_Y]).T\n    \n    return POSE_MEAN, POSE_STD\n\nPOSE_MEAN, POSE_STD = get_pose_mean_std()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:31.800030Z","iopub.execute_input":"2023-04-26T19:36:31.800684Z","iopub.status.idle":"2023-04-26T19:36:34.762966Z","shell.execute_reply.started":"2023-04-26T19:36:31.800640Z","shell.execute_reply":"2023-04-26T19:36:34.761779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Samples","metadata":{}},{"cell_type":"code","source":"# Custom sampler to get a batch containing N times all signs\ndef get_train_batch_all_signs(X, y, NON_EMPTY_FRAME_IDXS, n=BATCH_ALL_SIGNS_N):\n    # Arrays to store batch in\n    X_batch = np.zeros([NUM_CLASSES*n, INPUT_SIZE, N_COLS, N_DIMS], dtype=np.float32)\n    y_batch = np.arange(0, NUM_CLASSES, step=1/n, dtype=np.float32).astype(np.int64)\n    non_empty_frame_idxs_batch = np.zeros([NUM_CLASSES*n, INPUT_SIZE], dtype=np.float32)\n    \n    # Dictionary mapping ordinally encoded sign to corresponding sample indices\n    CLASS2IDXS = {}\n    for i in range(NUM_CLASSES):\n        CLASS2IDXS[i] = np.argwhere(y == i).squeeze().astype(np.int32)\n            \n    while True:\n        # Fill batch arrays\n        for i in range(NUM_CLASSES):\n            idxs = np.random.choice(CLASS2IDXS[i], n)\n            X_batch[i*n:(i+1)*n] = X[idxs]\n            non_empty_frame_idxs_batch[i*n:(i+1)*n] = NON_EMPTY_FRAME_IDXS[idxs]\n        \n        yield { 'frames': X_batch, 'non_empty_frame_idxs': non_empty_frame_idxs_batch }, y_batch","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:34.764911Z","iopub.execute_input":"2023-04-26T19:36:34.765594Z","iopub.status.idle":"2023-04-26T19:36:34.776718Z","shell.execute_reply.started":"2023-04-26T19:36:34.765545Z","shell.execute_reply":"2023-04-26T19:36:34.774944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code defines a generator function get_train_batch_all_signs to generate a specified number (n) of training batches of all sign language signs.\n\nThis function takes as input a sign language dataset (X and y), a non-empty frame index set (NON_EMPTY_FRAME_IDXS ), and the number of all sign language tokens in a batch (n), and produces a training batch of NUM_CLASSES * n samples. This training batch contains a frames dictionary and a non_empty_frame_idxs dictionary to store the sample's sign language frames and corresponding non-empty frame indices. The y_batch array contains the serial numbers of all sign language tokens. The main logic of the function is to loop through all sign language tokens, select n samples from each token, and add them to the batch array. The generator keeps looping through these samples so that the model can keep getting training data throughout the training process.","metadata":{}},{"cell_type":"code","source":"dummy_dataset = get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN)\nX_batch, y_batch = next(dummy_dataset)\n\nfor k, v in X_batch.items():\n    print(f'{k} shape: {v.shape}, dtype: {v.dtype}')\n\n# Batch shape/dtype\nprint(f'y_batch shape: {y_batch.shape}, dtype: {y_batch.dtype}')\n# Verify each batch contains each sign exactly N times\ndisplay(pd.Series(y_batch).value_counts().to_frame('Counts'))","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:34.778890Z","iopub.execute_input":"2023-04-26T19:36:34.779241Z","iopub.status.idle":"2023-04-26T19:36:34.865624Z","shell.execute_reply.started":"2023-04-26T19:36:34.779209Z","shell.execute_reply":"2023-04-26T19:36:34.864481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"What this code does is to randomly select BATCH_ALL_SIGNS_N samples from the X_train and y_train datasets to generate a training batch containing all sign language markers.\n\nThe generator function get_train_batch_all_signs will create an infinite loop generator that produces training batches containing all sign language tokens. For testing purposes, dummy_dataset is a batch obtained from this generator, which contains X_batch, y_batch and NON_EMPTY_FRAME_IDXS_TRAIN dictionaries, and a constant BATCH_ALL_SIGNS_N containing the number of all sign language tokens.\n\nIn the above code, X_batch is a dictionary which contains frames and non_empty_frame_idxs dictionaries whose shapes and data types are printed. Also, the shape and data type of the y_batch array are printed. Finally, use the pd.Series(y_batch).value_counts() function to verify that each sign language sign was included BATCH_ALL_SIGNS_N times.","metadata":{}},{"cell_type":"markdown","source":"# Model Config","metadata":{}},{"cell_type":"markdown","source":"This code defines some constants and variables for the machine learning model.\n\nLAYER_NORM_EPS is a constant that sets the epsilon value for layer normalization.\n\nLIPS_UNITS, HANDS_UNITS, POSE_UNITS and UNITS are variables used to set the number of dense layer units for keypoints, final embedding and transformer embedding size.\n\nNUM_BLOCKS and MLP_RATIO are variables that set the number of transformer blocks and MLP ratios.\n\nEMBEDDING_DROPOUT, MLP_DROPOUT_RATIO and CLASSIFIER_DROPOUT_RATIO are variables used to set dropout ratios for embeddings, MLPs and classifiers.\n\nINIT_HE_UNIFORM, INIT_GLOROT_UNIFORM and INIT_ZEROS are variables, initializers for setting weights.\n\nGELU is a variable used to set the activation function.\n\nThe last line prints the value of UNITS.","metadata":{}},{"cell_type":"code","source":"# Epsilon value for layer normalisation\nLAYER_NORM_EPS = 1e-6\n\n# Dense layer units for landmarks\nLIPS_UNITS = 384\nHANDS_UNITS = 384\nPOSE_UNITS = 384\n# final embedding and transformer embedding size\nUNITS = 512\n\n# Transformer\nNUM_BLOCKS = 2\nMLP_RATIO = 2\n\n# Dropout\nEMBEDDING_DROPOUT = 0.00\nMLP_DROPOUT_RATIO = 0.30\nCLASSIFIER_DROPOUT_RATIO = 0.10\n\n# Initiailizers\nINIT_HE_UNIFORM = tf.keras.initializers.he_uniform\nINIT_GLOROT_UNIFORM = tf.keras.initializers.glorot_uniform\nINIT_ZEROS = tf.keras.initializers.constant(0.0)\n# Activations\nGELU = tf.keras.activations.gelu\n\nprint(f'UNITS: {UNITS}')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:34.867162Z","iopub.execute_input":"2023-04-26T19:36:34.867627Z","iopub.status.idle":"2023-04-26T19:36:34.875581Z","shell.execute_reply.started":"2023-04-26T19:36:34.867587Z","shell.execute_reply":"2023-04-26T19:36:34.874426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transformers \n\nNeed to implement transformer from scratch as TFLite does not support the native TF implementation of MultiHeadAttention.由于TFLite（TensorFlow Lite，一。A version of TensorFlow optimized for mobile and embedded devices) does not support TensorFlow's native MultiHeadAttention layer implementation, so the Transformer model (a neural network architecture used in natural language processing tasks) needs to be implemented from scratch, including the MultiHeadAttention layer. Since the MultiHeadAttention layer is a key component of the Transformer model, it may not be possible to use a pre-existing Transformer implementation in TensorFlow in TFLite if its implementation is not supported by TFLite. Therefore, you need to write the implementation code of Transformer yourself instead of relying on the implementation of the MultiHeadAttention layer in TensorFlow","metadata":{}},{"cell_type":"markdown","source":"Encoder: The input is the Embedding of the word, plus the position encoding, and then enters a unified structure, which can be looped many times (N times), that is to say, there are many layers (N layers). Each layer can be divided into an Attention layer and a fully connected layer, and some additional processing is added, such as Skip Connection, to make a skip connection, and then a Normalization layer is added. In fact, its own model is still very simple. Decoder: The first input is the prefix information, and the subsequent one is the Embedding produced last time, adding the position code, and then entering a module that can be repeated many times. This module can be divided into three parts. The first part is also the Attention layer, the second part is cross Attention, not Self-Attention, and the third part is the fully connected layer. Also used skip connection and Normalization. Output: The final output must pass through the Linear layer (full connection layer), and then predict through softmax.\n\nadd Codeadd Markdown\nThe Encoder part is a stack of N identical structures, and each structure can be subdivided into the following structures:\n\n1. Perform embedding (word embedding) on ​​the input one-hot encoded samples\n2. Add location code\n3. Introduce self-attention of multi-head mechanism\n4. Add the input and output of self-attention (residual network structure)\n5. Layer Normalization (layer normalization), normalize the data at all times\n6. Feedforword neural network structure\n7. Add the input and output of Feedforword (residual network structure)\n8. Layer Normalization, standardize the data at all times\n9. Repeat the structure of N layers 3-8\n10. add Codeadd Markdown\nThe Decoder part is also a stack of N identical structures, and each structure can be subdivided into the following structures:\n\n1. Perform embedding (word embedding) on ​​the input one-hot encoded samples\n2. Add location code\n3. Introduce self-attention of multi-head mechanism\n4. Add the input and output of self-attention (residual network structure)\n5. Layer Normalization,\n6. Standardize the data at all times and use the value obtained in the previous step as the value, and perform Self-Attenton with q and k obtained from the encoder\n7. Add the input and output of self-attention (residual network structure)\n8. Layer Normalization (layer normalization), normalize the data at all times\n9. Feedforword neural network (Feedforword) structure\n10. Add the input and output of Feedforword (residual network structure)\n11. Layer Normalization, standardize the data at all times\n12. Repeat the structure of N layer 3-11","metadata":{}},{"cell_type":"code","source":"# based on: https://stackoverflow.com/questions/67342988/verifying-the-implementation-of-multihead-attention-in-transformer\n# replaced softmax with softmax layer to support masked softmax\ndef scaled_dot_product(q,k,v, softmax, attention_mask):\n    #calculates Q . K(transpose)\n    qkt = tf.matmul(q,k,transpose_b=True)\n    #caculates scaling factor\n    dk = tf.math.sqrt(tf.cast(q.shape[-1],dtype=tf.float32))\n    scaled_qkt = qkt/dk\n    softmax = softmax(scaled_qkt, mask=attention_mask)\n    \n    z = tf.matmul(softmax,v)\n    #shape: (m,Tx,depth), same shape as q,k,v\n    return z\n\nclass MultiHeadAttention(tf.keras.layers.Layer):\n    def __init__(self,d_model,num_of_heads):\n        super(MultiHeadAttention,self).__init__()\n        self.d_model = d_model\n        self.num_of_heads = num_of_heads\n        self.depth = d_model//num_of_heads\n        self.wq = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wk = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wv = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wo = tf.keras.layers.Dense(d_model)\n        self.softmax = tf.keras.layers.Softmax()\n        \n    def call(self,x, attention_mask):\n        \n        multi_attn = []\n        for i in range(self.num_of_heads):\n            Q = self.wq[i](x)\n            K = self.wk[i](x)\n            V = self.wv[i](x)\n            multi_attn.append(scaled_dot_product(Q,K,V, self.softmax, attention_mask))\n            \n        multi_head = tf.concat(multi_attn,axis=-1)\n        multi_head_attention = self.wo(multi_head)\n        return multi_head_attention","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:34.877688Z","iopub.execute_input":"2023-04-26T19:36:34.878151Z","iopub.status.idle":"2023-04-26T19:36:34.892772Z","shell.execute_reply.started":"2023-04-26T19:36:34.878111Z","shell.execute_reply":"2023-04-26T19:36:34.891647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code is an implementation of MultiHeadAttention. It linearly transforms the input tensor x through multiple Dense layers, and then passes the transformed tensors into the scaled_dot_product function as Q, K, and V respectively, and calculates the output of the multi-head attention mechanism. Finally, the output of the multi-head attention mechanism is spliced ​​together, and then linearly transformed through a Dense layer to obtain the final output multi_head_attention. The scaled_dot_product function is a function to calculate Q.K^T, where Q, K, and V are query, key, and value matrices respectively, and attention_mask is a tensor used for the mask. The softmax function is a function used to calculate the softmax value.","metadata":{}},{"cell_type":"code","source":"# Full Transformer\nclass Transformer(tf.keras.Model):\n    def __init__(self, num_blocks):\n        super(Transformer, self).__init__(name='transformer')\n        self.num_blocks = num_blocks\n    \n    def build(self, input_shape):\n        self.ln_1s = []\n        self.mhas = []\n        self.ln_2s = []\n        self.mlps = []\n        # Make Transformer Blocks\n        for i in range(self.num_blocks):\n            # Multi Head Attention\n            self.mhas.append(MultiHeadAttention(UNITS, 8))\n            # Multi Layer Perception\n            self.mlps.append(tf.keras.Sequential([\n                tf.keras.layers.Dense(UNITS * MLP_RATIO, activation=GELU, kernel_initializer=INIT_GLOROT_UNIFORM),\n                tf.keras.layers.Dropout(MLP_DROPOUT_RATIO),\n                tf.keras.layers.Dense(UNITS, kernel_initializer=INIT_HE_UNIFORM),\n            ]))\n        \n    def call(self, x, attention_mask):\n        # Iterate input over transformer blocks\n        for mha, mlp in zip(self.mhas, self.mlps):\n            x = x + mha(x, attention_mask)\n            x = x + mlp(x)\n    \n        return x","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:34.894448Z","iopub.execute_input":"2023-04-26T19:36:34.894835Z","iopub.status.idle":"2023-04-26T19:36:34.907405Z","shell.execute_reply.started":"2023-04-26T19:36:34.894786Z","shell.execute_reply":"2023-04-26T19:36:34.906372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is a Python Transformer model, which is a class inherited from tf.keras.Model. It has a constructor where num_blocks is an integer representing the number of Transformer blocks. In the build function, it creates multiple Multi Head Attention and Multi Layer Perception objects and stores them in class variables. In the call function, it iterates over the input data and passes it to each Transformer block. Each block contains a Multi Head Attention and a Multi Layer Perception layer. The purpose of this model is to implement natural language processing tasks, such as machine translation, text summarization, etc.","metadata":{}},{"cell_type":"markdown","source":"# Landmark Embidding\n\nKey point embedding\", where \"Landmark\" represents the key points of the face, and \"Embedding\" represents the process of mapping these key point information to a low-dimensional vector space. Therefore, the Chinese meaning of \"Landmark Embedding\" can be understood as \"the face Embedding key point information into low-dimensional vector space\".\n\nadd Codeadd Markdown\nLandmark Embedding It is a method to convert face key point information into a low-dimensional vector representation. In tasks such as face recognition and facial expression recognition, Landmark Embedding is usually used to extract facial feature representations.\n\nSpecifically, Landmark Embedding maps the coordinates of key points in a face image to a vector representation in a low-dimensional space. This vector representation can contain information about face shape, pose, and expression, and can be used to compare similarities or differences between different faces. Compared with directly using pixel information or high-dimensional feature vector representation, Landmark Embedding can improve the accuracy and robustness of face recognition and expression recognition.","metadata":{}},{"cell_type":"code","source":"class LandmarkEmbedding(tf.keras.Model):\n    def __init__(self, units, name):\n        super(LandmarkEmbedding, self).__init__(name=f'{name}_embedding')\n        self.units = units\n        \n    def build(self, input_shape):\n        # Embedding for missing landmark in frame, initizlied with zeros\n        self.empty_embedding = self.add_weight(\n            name=f'{self.name}_empty_embedding',\n            shape=[self.units],\n            initializer=INIT_ZEROS,\n        )\n        # Embedding\n        self.dense = tf.keras.Sequential([\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM),\n            tf.keras.layers.Activation(GELU),\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name=f'{self.name}_dense')\n\n    def call(self, x):\n        return tf.where(\n                # Checks whether landmark is missing in frame\n                tf.reduce_sum(x, axis=2, keepdims=True) == 0,\n                # If so, the empty embedding is used\n                self.empty_embedding,\n                # Otherwise the landmark data is embedded\n                self.dense(x),\n            )","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:34.909049Z","iopub.execute_input":"2023-04-26T19:36:34.909777Z","iopub.status.idle":"2023-04-26T19:36:34.920106Z","shell.execute_reply.started":"2023-04-26T19:36:34.909736Z","shell.execute_reply":"2023-04-26T19:36:34.919376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Embeddding**","metadata":{}},{"cell_type":"code","source":"class Embedding(tf.keras.Model):\n    def __init__(self):\n        super(Embedding, self).__init__()\n        \n    def get_diffs(self, l):\n        S = l.shape[2]\n        other = tf.expand_dims(l, 3)\n        other = tf.repeat(other, S, axis=3)\n        other = tf.transpose(other, [0,1,3,2])\n        diffs = tf.expand_dims(l, 3) - other\n        diffs = tf.reshape(diffs, [-1, INPUT_SIZE, S*S])\n        return diffs\n\n    def build(self, input_shape):\n        # Positional Embedding, initialized with zeros\n        self.positional_embedding = tf.keras.layers.Embedding(INPUT_SIZE+1, UNITS, embeddings_initializer=INIT_ZEROS)\n        # Embedding layer for Landmarks\n        self.lips_embedding = LandmarkEmbedding(LIPS_UNITS, 'lips')\n        self.left_hand_embedding = LandmarkEmbedding(HANDS_UNITS, 'left_hand')\n        self.pose_embedding = LandmarkEmbedding(POSE_UNITS, 'pose')\n        # Landmark Weights\n        self.landmark_weights = tf.Variable(tf.zeros([3], dtype=tf.float32), name='landmark_weights')\n        # Fully Connected Layers for combined landmarks\n        self.fc = tf.keras.Sequential([\n            tf.keras.layers.Dense(UNITS, name='fully_connected_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM),\n            tf.keras.layers.Activation(GELU),\n            tf.keras.layers.Dense(UNITS, name='fully_connected_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name='fc')\n\n\n    def call(self, lips0, left_hand0, pose0, non_empty_frame_idxs, training=False):\n        # Lips\n        lips_embedding = self.lips_embedding(lips0)\n        # Left Hand\n        left_hand_embedding = self.left_hand_embedding(left_hand0)\n        # Pose\n        pose_embedding = self.pose_embedding(pose0)\n        # Merge Embeddings of all landmarks with mean pooling\n        x = tf.stack((\n            lips_embedding, left_hand_embedding, pose_embedding,\n        ), axis=3)\n        x = x * tf.nn.softmax(self.landmark_weights)\n        x = tf.reduce_sum(x, axis=3)\n        # Fully Connected Layers\n        x = self.fc(x)\n        # Add Positional Embedding\n        max_frame_idxs = tf.clip_by_value(\n                tf.reduce_max(non_empty_frame_idxs, axis=1, keepdims=True),\n                1,\n                np.PINF,\n            )\n        normalised_non_empty_frame_idxs = tf.where(\n            tf.math.equal(non_empty_frame_idxs, -1.0),\n            INPUT_SIZE,\n            tf.cast(\n                non_empty_frame_idxs / max_frame_idxs * INPUT_SIZE,\n                tf.int32,\n            ),\n        )\n        x = x + self.positional_embedding(normalised_non_empty_frame_idxs)\n        \n        return x","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:34.921650Z","iopub.execute_input":"2023-04-26T19:36:34.922420Z","iopub.status.idle":"2023-04-26T19:36:34.937671Z","shell.execute_reply.started":"2023-04-26T19:36:34.922379Z","shell.execute_reply":"2023-04-26T19:36:34.936584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Augmentation \n\nThe function of the add_noise method is to replace all 0 in the input tensor t with 0, and add a random noise that obeys a normal distribution to all non-zero elements. The standard deviation of this noise is noise_std, which is defined in the constructor of the class. This method uses TensorFlow's tf.where function, which takes three arguments: a boolean tensor, a tensor x, and a tensor y. If the element in the Boolean tensor is True, return the element at the corresponding position in x; otherwise return the element at the corresponding position in y. In this method, if the element in t is 0, it returns 0; otherwise, it returns t plus a normal distribution of random noise. If the train flag is True, noise will be added on each tensor.","metadata":{}},{"cell_type":"code","source":"# Not used, adds random X/y translation to input on samples level\nclass Augmentation(tf.keras.layers.Layer):\n    def __init__(self, noise_std):\n        super(Augmentation, self).__init__()\n        self.noise_std = noise_std\n    \n    def add_noise(self, t):\n        B = tf.shape(t)[0]\n        return tf.where(\n            t == 0.0,\n            0.0,\n            t + tf.random.normal([B,1,1,tf.shape(t)[3]], 0, self.noise_std),\n        )\n    \n    def call(self, lips0, left_hand0, pose0, training=False):\n        if training:\n            # Lips\n            lips0 = self.add_noise(lips0)\n            # Left Hand\n            left_hand0 = self.add_noise(left_hand0)\n            # Pose\n            pose0 = self.add_noise(pose0)\n        \n        return lips0, left_hand0, pose0","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:34.939048Z","iopub.execute_input":"2023-04-26T19:36:34.939520Z","iopub.status.idle":"2023-04-26T19:36:34.950878Z","shell.execute_reply.started":"2023-04-26T19:36:34.939472Z","shell.execute_reply":"2023-04-26T19:36:34.949783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sparse Categorical Crossentropy With Label Smoothing","metadata":{}},{"cell_type":"code","source":"# source:: https://stackoverflow.com/questions/60689185/label-smoothing-for-sparse-categorical-crossentropy\ndef scce_with_ls(y_true, y_pred):\n    # One Hot Encode Sparsely Encoded Target Sign\n    y_true = tf.cast(y_true, tf.int32)\n    y_true = tf.one_hot(y_true, NUM_CLASSES, axis=1)\n    y_true = tf.squeeze(y_true, axis=2)\n    # Categorical Crossentropy with native label smoothing support\n    return tf.keras.losses.categorical_crossentropy(y_true, y_pred, label_smoothing=0.25)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:34.954238Z","iopub.execute_input":"2023-04-26T19:36:34.954525Z","iopub.status.idle":"2023-04-26T19:36:34.966241Z","shell.execute_reply.started":"2023-04-26T19:36:34.954497Z","shell.execute_reply":"2023-04-26T19:36:34.965107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model\nThis code is an implementation of a TensorFlow model. It has two inputs, \"frames\" and \"non_empty_frame_idxs\". In this model, frames is a video data containing multiple frames, and non_empty_frame_idxs indicates which frames in frames have content. In this code, locations with valid frames are selected by masking so that only those frames are trained. This model uses the Transformer architecture, which processes input by combining multiple layers with attention mechanisms. In this model, each frame is embedded into three different representations, namely LIPS, LEFT HAND and POSE, which are used to construct the input of the Transformer. In this model, some additional tricks are also implemented, such as random frame masking, category loss (some features are lost during classification), and label smoothing, etc. Finally, the model also includes an optimizer and some evaluation metrics. The optimizer uses AdamW, and the evaluation metrics include sparse classification accuracy, top-k accuracy for sparse classification, etc","metadata":{}},{"cell_type":"code","source":"def get_model():\n    # Inputs\n    frames = tf.keras.layers.Input([INPUT_SIZE, N_COLS, N_DIMS], dtype=tf.float32, name='frames')\n    non_empty_frame_idxs = tf.keras.layers.Input([INPUT_SIZE], dtype=tf.float32, name='non_empty_frame_idxs')\n    # Padding Mask\n    mask0 = tf.cast(tf.math.not_equal(non_empty_frame_idxs, -1), tf.float32)\n    mask0 = tf.expand_dims(mask0, axis=2)\n    # Random Frame Masking\n    mask = tf.where(\n        (tf.random.uniform(tf.shape(mask0)) > 0.25) & tf.math.not_equal(mask0, 0.0),\n        1.0,\n        0.0,\n    )\n    # Correct Samples Which are all masked now...\n    mask = tf.where(\n        tf.math.equal(tf.reduce_sum(mask, axis=[1,2], keepdims=True), 0.0),\n        mask0,\n        mask,\n    )\n    \n    \n    \"\"\"\n        left_hand: 468:489\n        pose: 489:522\n        right_hand: 522:543\n    \"\"\"\n    x = frames\n    x = tf.slice(x, [0,0,0,0], [-1,INPUT_SIZE, N_COLS, 2])\n    # LIPS\n    lips = tf.slice(x, [0,0,LIPS_START,0], [-1,INPUT_SIZE, 40, 2])\n    lips = tf.where(\n            tf.math.equal(lips, 0.0),\n            0.0,\n            (lips - LIPS_MEAN) / LIPS_STD,\n        )\n    # LEFT HAND\n    left_hand = tf.slice(x, [0,0,40,0], [-1,INPUT_SIZE, 21, 2])\n    left_hand = tf.where(\n            tf.math.equal(left_hand, 0.0),\n            0.0,\n            (left_hand - LEFT_HANDS_MEAN) / LEFT_HANDS_STD,\n        )\n    # POSE\n    pose = tf.slice(x, [0,0,61,0], [-1,INPUT_SIZE, 5, 2])\n    pose = tf.where(\n            tf.math.equal(pose, 0.0),\n            0.0,\n            (pose - POSE_MEAN) / POSE_STD,\n        )\n    \n    # Flatten\n    lips = tf.reshape(lips, [-1, INPUT_SIZE, 40*2])\n    left_hand = tf.reshape(left_hand, [-1, INPUT_SIZE, 21*2])\n    pose = tf.reshape(pose, [-1, INPUT_SIZE, 5*2])\n        \n    # Embedding\n    x = Embedding()(lips, left_hand, pose, non_empty_frame_idxs)\n    \n    # Encoder Transformer Blocks\n    x = Transformer(NUM_BLOCKS)(x, mask)\n    \n    # Pooling\n    x = tf.reduce_sum(x * mask, axis=1) / tf.reduce_sum(mask, axis=1)\n    # Classifier Dropout\n    x = tf.keras.layers.Dropout(CLASSIFIER_DROPOUT_RATIO)(x)\n    # Classification Layer\n    x = tf.keras.layers.Dense(NUM_CLASSES, activation=tf.keras.activations.softmax, kernel_initializer=INIT_GLOROT_UNIFORM)(x)\n    \n    outputs = x\n    \n    # Create Tensorflow Model\n    model = tf.keras.models.Model(inputs=[frames, non_empty_frame_idxs], outputs=outputs)\n    \n    # Sparse Categorical Cross Entropy With Label Smoothing\n    loss = scce_with_ls\n    \n    # Adam Optimizer with weight decay\n    optimizer = tfa.optimizers.AdamW(learning_rate=1e-3, weight_decay=1e-5, clipnorm=1.0)\n    \n    # TopK Metrics\n    metrics = [\n        tf.keras.metrics.SparseCategoricalAccuracy(name='acc'),\n        tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5, name='top_5_acc'),\n        tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10, name='top_10_acc'),\n    ]\n    \n    model.compile(loss=loss, optimizer=optimizer, metrics=metrics)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:34.967732Z","iopub.execute_input":"2023-04-26T19:36:34.968420Z","iopub.status.idle":"2023-04-26T19:36:34.987245Z","shell.execute_reply.started":"2023-04-26T19:36:34.968382Z","shell.execute_reply":"2023-04-26T19:36:34.986492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nmodel = get_model()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:34.989335Z","iopub.execute_input":"2023-04-26T19:36:34.989975Z","iopub.status.idle":"2023-04-26T19:36:37.362056Z","shell.execute_reply.started":"2023-04-26T19:36:34.989927Z","shell.execute_reply":"2023-04-26T19:36:37.360915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot model summary\nmodel.summary(expand_nested=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:37.363795Z","iopub.execute_input":"2023-04-26T19:36:37.364166Z","iopub.status.idle":"2023-04-26T19:36:37.528188Z","shell.execute_reply.started":"2023-04-26T19:36:37.364126Z","shell.execute_reply":"2023-04-26T19:36:37.527336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(model, show_shapes=True, show_dtype=True, show_layer_names=True, expand_nested=True, show_layer_activations=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:37.548520Z","iopub.execute_input":"2023-04-26T19:36:37.548895Z","iopub.status.idle":"2023-04-26T19:36:38.934220Z","shell.execute_reply.started":"2023-04-26T19:36:37.548856Z","shell.execute_reply":"2023-04-26T19:36:38.933108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# no NaN predictions","metadata":{}},{"cell_type":"code","source":"if not PREPROCESS_DATA and TRAIN_MODEL:\n    y_pred = model.predict_on_batch(X_batch).flatten()\n\n    print(f'# NaN Values In Prediction: {np.isnan(y_pred).sum()}')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:38.936120Z","iopub.execute_input":"2023-04-26T19:36:38.937371Z","iopub.status.idle":"2023-04-26T19:36:43.259420Z","shell.execute_reply.started":"2023-04-26T19:36:38.937330Z","shell.execute_reply":"2023-04-26T19:36:43.257067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weight Predictions","metadata":{}},{"cell_type":"code","source":"if not PREPROCESS_DATA and TRAIN_MODEL:\n    plt.figure(figsize=(12,5))\n    plt.title(f'Softmax Output Initialized Model | µ={y_pred.mean():.3f}, σ={y_pred.std():.3f}', pad=25)\n    pd.Series(y_pred).plot(kind='hist', bins=128, label='Class Probability')\n    plt.xlim(0, max(y_pred) * 1.1)\n    plt.vlines([1 / NUM_CLASSES], 0, plt.ylim()[1], color='red', label=f'Random Guessing Baseline 1/NUM_CLASSES={1 / NUM_CLASSES:.3f}')\n    plt.grid()\n    plt.legend()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:43.261246Z","iopub.execute_input":"2023-04-26T19:36:43.262877Z","iopub.status.idle":"2023-04-26T19:36:43.884607Z","shell.execute_reply.started":"2023-04-26T19:36:43.262822Z","shell.execute_reply":"2023-04-26T19:36:43.883543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Learning rate scheduler","metadata":{}},{"cell_type":"code","source":"def lrfn(current_step, num_warmup_steps, lr_max, num_cycles=0.50, num_training_steps=N_EPOCHS):\n    \n    if current_step < num_warmup_steps:\n        if WARMUP_METHOD == 'log':\n            return lr_max * 0.10 ** (num_warmup_steps - current_step)\n        else:\n            return lr_max * 2 ** -(num_warmup_steps - current_step)\n    else:\n        progress = float(current_step - num_warmup_steps) / float(max(1, num_training_steps - num_warmup_steps))\n\n        return max(0.0, 0.5 * (1.0 + math.cos(math.pi * float(num_cycles) * 2.0 * progress))) * lr_max","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:43.886377Z","iopub.execute_input":"2023-04-26T19:36:43.887127Z","iopub.status.idle":"2023-04-26T19:36:43.895099Z","shell.execute_reply.started":"2023-04-26T19:36:43.887085Z","shell.execute_reply":"2023-04-26T19:36:43.894074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_lr_schedule(lr_schedule, epochs):\n    fig = plt.figure(figsize=(20, 10))\n    plt.plot([None] + lr_schedule + [None])\n    # X Labels\n    x = np.arange(1, epochs + 1)\n    x_axis_labels = [i if epochs <= 40 or i % 5 == 0 or i == 1 else None for i in range(1, epochs + 1)]\n    plt.xlim([1, epochs])\n    plt.xticks(x, x_axis_labels) # set tick step to 1 and let x axis start at 1\n    \n    # Increase y-limit for better readability\n    plt.ylim([0, max(lr_schedule) * 1.1])\n    \n    # Title\n    schedule_info = f'start: {lr_schedule[0]:.1E}, max: {max(lr_schedule):.1E}, final: {lr_schedule[-1]:.1E}'\n    plt.title(f'Step Learning Rate Schedule, {schedule_info}', size=18, pad=12)\n    \n    # Plot Learning Rates\n    for x, val in enumerate(lr_schedule):\n        if epochs <= 40 or x % 5 == 0 or x is epochs - 1:\n            if x < len(lr_schedule) - 1:\n                if lr_schedule[x - 1] < val:\n                    ha = 'right'\n                else:\n                    ha = 'left'\n            elif x == 0:\n                ha = 'right'\n            else:\n                ha = 'left'\n            plt.plot(x + 1, val, 'o', color='black');\n            offset_y = (max(lr_schedule) - min(lr_schedule)) * 0.02\n            plt.annotate(f'{val:.1E}', xy=(x + 1, val + offset_y), size=12, ha=ha)\n    \n    plt.xlabel('Epoch', size=16, labelpad=5)\n    plt.ylabel('Learning Rate', size=16, labelpad=5)\n    plt.grid()\n    plt.show()\n\n# Learning rate for encoder\nLR_SCHEDULE = [lrfn(step, num_warmup_steps=N_WARMUP_EPOCHS, lr_max=LR_MAX, num_cycles=0.50) for step in range(N_EPOCHS)]\n# Plot Learning Rate Schedule\nplot_lr_schedule(LR_SCHEDULE, epochs=N_EPOCHS)\n# Learning Rate Callback\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lambda step: LR_SCHEDULE[step], verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:43.896690Z","iopub.execute_input":"2023-04-26T19:36:43.897304Z","iopub.status.idle":"2023-04-26T19:36:44.680476Z","shell.execute_reply.started":"2023-04-26T19:36:43.897261Z","shell.execute_reply":"2023-04-26T19:36:44.676602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weight Decay Callback","metadata":{}},{"cell_type":"code","source":"# Custom callback to update weight decay with learning rate\nclass WeightDecayCallback(tf.keras.callbacks.Callback):\n    def __init__(self, wd_ratio=WD_RATIO):\n        self.step_counter = 0\n        self.wd_ratio = wd_ratio\n    \n    def on_epoch_begin(self, epoch, logs=None):\n        model.optimizer.weight_decay = model.optimizer.learning_rate * self.wd_ratio\n        print(f'learning rate: {model.optimizer.learning_rate.numpy():.2e}, weight decay: {model.optimizer.weight_decay.numpy():.2e}')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:44.683992Z","iopub.execute_input":"2023-04-26T19:36:44.684298Z","iopub.status.idle":"2023-04-26T19:36:44.690253Z","shell.execute_reply.started":"2023-04-26T19:36:44.684263Z","shell.execute_reply":"2023-04-26T19:36:44.689184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Performance benchmark","metadata":{}},{"cell_type":"code","source":"%%timeit -n 100\nif TRAIN_MODEL:\n    # Verify model prediction is <<<100ms\n    model.predict_on_batch({ 'frames': X_train[:1], 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_TRAIN[:1] })\n    pass","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:44.692223Z","iopub.execute_input":"2023-04-26T19:36:44.692997Z","iopub.status.idle":"2023-04-26T19:36:59.207023Z","shell.execute_reply.started":"2023-04-26T19:36:44.692953Z","shell.execute_reply":"2023-04-26T19:36:59.205775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"if USE_VAL:\n    # Verify Validation Dataset Covers All Signs\n    print(f'# Unique Signs in Validation Set: {pd.Series(y_val).nunique()}')\n    # Value Counts\n    display(pd.Series(y_val).value_counts().to_frame('Count').iloc[[1,2,3,-3,-2,-1]])","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:59.208342Z","iopub.execute_input":"2023-04-26T19:36:59.208997Z","iopub.status.idle":"2023-04-26T19:36:59.215909Z","shell.execute_reply.started":"2023-04-26T19:36:59.208951Z","shell.execute_reply":"2023-04-26T19:36:59.214716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluate Initialized Model","metadata":{}},{"cell_type":"code","source":"# Sanity Check\nif TRAIN_MODEL and USE_VAL:\n    _ = model.evaluate(*validation_data, verbose=2)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:59.217743Z","iopub.execute_input":"2023-04-26T19:36:59.218576Z","iopub.status.idle":"2023-04-26T19:36:59.226422Z","shell.execute_reply.started":"2023-04-26T19:36:59.218517Z","shell.execute_reply":"2023-04-26T19:36:59.225574Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"if TRAIN_MODEL:\n    # Clear all models in GPU\n    tf.keras.backend.clear_session()\n\n    # Get new fresh model\n    model = get_model()\n    \n    # Sanity Check\n    model.summary()\n\n    # Actual Training\n    history = model.fit(\n            x=get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN),\n            steps_per_epoch=len(X_train) // (NUM_CLASSES * BATCH_ALL_SIGNS_N),\n            epochs=N_EPOCHS,\n            # Only used for validation data since training data is a generator\n            batch_size=BATCH_SIZE,\n            validation_data=validation_data,\n            callbacks=[\n                lr_callback,\n                WeightDecayCallback(),\n            ],\n            verbose = VERBOSE,\n        )","metadata":{"execution":{"iopub.status.busy":"2023-04-26T19:36:59.228371Z","iopub.execute_input":"2023-04-26T19:36:59.229346Z","iopub.status.idle":"2023-04-26T21:49:35.715950Z","shell.execute_reply.started":"2023-04-26T19:36:59.229304Z","shell.execute_reply":"2023-04-26T21:49:35.714738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"model.fit()is a method in TensorFlow for training a model on a given dataset. It accepts multiple parameters such as training data, validation data, number of epochs, batch size, etc. It trains the model by using an optimizer to minimize a loss function. A loss function is a way to measure how well a model performs at predicting an output. The optimizer adjusts the weights of the model to minimize this loss function. During training, model.fit() prints out metrics like loss and accuracy","metadata":{}},{"cell_type":"code","source":"# Save Model Weights\nmodel.save_weights('model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T21:49:35.717759Z","iopub.execute_input":"2023-04-26T21:49:35.718705Z","iopub.status.idle":"2023-04-26T21:49:35.905906Z","shell.execute_reply.started":"2023-04-26T21:49:35.718658Z","shell.execute_reply":"2023-04-26T21:49:35.904784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if USE_VAL:\n    # Validation Predictions\n    y_val_pred = model.predict({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, verbose=2).argmax(axis=1)\n    # Label\n    labels = [ORD2SIGN.get(i).replace(' ', '_') for i in range(NUM_CLASSES)]","metadata":{"execution":{"iopub.status.busy":"2023-04-26T22:17:38.562237Z","iopub.execute_input":"2023-04-26T22:17:38.562676Z","iopub.status.idle":"2023-04-26T22:17:38.568999Z","shell.execute_reply.started":"2023-04-26T22:17:38.562635Z","shell.execute_reply":"2023-04-26T22:17:38.567710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Landmark Attention Weigh","metadata":{}},{"cell_type":"code","source":"# Landmark Weights\nfor w in model.get_layer('embedding').weights:\n    if 'landmark_weights' in w.name:\n        weights = scipy.special.softmax(w)\n\nlandmarks = ['lips_embedding', 'left_hand_embedding', 'pose_embedding']\n\nfor w, lm in zip(weights, landmarks):\n    print(f'{lm} weight: {(w*100):.1f}%')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T21:49:35.917133Z","iopub.status.idle":"2023-04-26T21:49:35.917973Z","shell.execute_reply.started":"2023-04-26T21:49:35.917667Z","shell.execute_reply":"2023-04-26T21:49:35.917717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Classification Report","metadata":{}},{"cell_type":"code","source":"def print_classification_report():\n    # Classification report for all signs\n    classification_report = sklearn.metrics.classification_report(\n            y_val,\n            y_val_pred,\n            target_names=labels,\n            output_dict=True,\n        )\n    # Round Data for better readability\n    classification_report = pd.DataFrame(classification_report).T\n    classification_report = classification_report.round(2)\n    classification_report = classification_report.astype({\n            'support': np.uint16,\n        })\n    # Add signs\n    classification_report['sign'] = [e if e in SIGN2ORD else -1 for e in classification_report.index]\n    classification_report['sign_ord'] = classification_report['sign'].apply(SIGN2ORD.get).fillna(-1).astype(np.int16)\n    # Sort on F1-score\n    classification_report = pd.concat((\n        classification_report.head(NUM_CLASSES).sort_values('f1-score', ascending=False),\n        classification_report.tail(3),\n    ))\n\n    pd.options.display.max_rows = 999\n    display(classification_report)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T21:49:35.919470Z","iopub.status.idle":"2023-04-26T21:49:35.920248Z","shell.execute_reply.started":"2023-04-26T21:49:35.919981Z","shell.execute_reply":"2023-04-26T21:49:35.920009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if USE_VAL:\n    print_classification_report()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T22:17:50.441677Z","iopub.execute_input":"2023-04-26T22:17:50.442084Z","iopub.status.idle":"2023-04-26T22:17:50.447177Z","shell.execute_reply.started":"2023-04-26T22:17:50.442048Z","shell.execute_reply":"2023-04-26T22:17:50.446025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training History","metadata":{}},{"cell_type":"code","source":"def plot_history_metric(metric, f_best=np.argmax, ylim=None, yscale=None, yticks=None):\n    plt.figure(figsize=(20, 10))\n    \n    values = history.history[metric]\n    N_EPOCHS = len(values)\n    val = 'val' in ''.join(history.history.keys())\n    # Epoch Ticks\n    if N_EPOCHS <= 20:\n        x = np.arange(1, N_EPOCHS + 1)\n    else:\n        x = [1, 5] + [10 + 5 * idx for idx in range((N_EPOCHS - 10) // 5 + 1)]\n\n    x_ticks = np.arange(1, N_EPOCHS+1)\n\n    # Validation\n    if val:\n        val_values = history.history[f'val_{metric}']\n        val_argmin = f_best(val_values)\n        plt.plot(x_ticks, val_values, label=f'val')\n\n    # summarize history for accuracy\n    plt.plot(x_ticks, values, label=f'train')\n    argmin = f_best(values)\n    plt.scatter(argmin + 1, values[argmin], color='red', s=75, marker='o', label=f'train_best')\n    if val:\n        plt.scatter(val_argmin + 1, val_values[val_argmin], color='purple', s=75, marker='o', label=f'val_best')\n\n    plt.title(f'Model {metric}', fontsize=24, pad=10)\n    plt.ylabel(metric, fontsize=20, labelpad=10)\n\n    if ylim:\n        plt.ylim(ylim)\n\n    if yscale is not None:\n        plt.yscale(yscale)\n        \n    if yticks is not None:\n        plt.yticks(yticks, fontsize=16)\n\n    plt.xlabel('epoch', fontsize=20, labelpad=10)        \n    plt.tick_params(axis='x', labelsize=8)\n    plt.xticks(x, fontsize=16) # set tick step to 1 and let x axis start at 1\n    plt.yticks(fontsize=16)\n    \n    plt.legend(prop={'size': 10})\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-26T22:18:12.324357Z","iopub.execute_input":"2023-04-26T22:18:12.325121Z","iopub.status.idle":"2023-04-26T22:18:12.338779Z","shell.execute_reply.started":"2023-04-26T22:18:12.325077Z","shell.execute_reply":"2023-04-26T22:18:12.337559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('loss', f_best=np.argmin)","metadata":{"execution":{"iopub.status.busy":"2023-04-26T22:18:16.004745Z","iopub.execute_input":"2023-04-26T22:18:16.005753Z","iopub.status.idle":"2023-04-26T22:18:16.422408Z","shell.execute_reply.started":"2023-04-26T22:18:16.005711Z","shell.execute_reply":"2023-04-26T22:18:16.421228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1))","metadata":{"execution":{"iopub.status.busy":"2023-04-26T22:18:21.051475Z","iopub.execute_input":"2023-04-26T22:18:21.052247Z","iopub.status.idle":"2023-04-26T22:18:21.479590Z","shell.execute_reply.started":"2023-04-26T22:18:21.052190Z","shell.execute_reply":"2023-04-26T22:18:21.478500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('top_5_acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1))","metadata":{"execution":{"iopub.status.busy":"2023-04-26T22:18:24.370101Z","iopub.execute_input":"2023-04-26T22:18:24.371187Z","iopub.status.idle":"2023-04-26T22:18:24.776967Z","shell.execute_reply.started":"2023-04-26T22:18:24.371132Z","shell.execute_reply":"2023-04-26T22:18:24.775954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('top_10_acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1))","metadata":{"execution":{"iopub.status.busy":"2023-04-26T22:18:27.891778Z","iopub.execute_input":"2023-04-26T22:18:27.894199Z","iopub.status.idle":"2023-04-26T22:18:28.304639Z","shell.execute_reply.started":"2023-04-26T22:18:27.894148Z","shell.execute_reply":"2023-04-26T22:18:28.303476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"# TFLite model for submission\nclass TFLiteModel(tf.Module):\n    def __init__(self, model):\n        super(TFLiteModel, self).__init__()\n\n        # Load the feature generation and main models\n        self.preprocess_layer = preprocess_layer\n        self.model = model\n    \n    @tf.function(input_signature=[tf.TensorSpec(shape=[None, N_ROWS, N_DIMS], dtype=tf.float32, name='inputs')])\n    def __call__(self, inputs):\n        # Preprocess Data\n        x, non_empty_frame_idxs = self.preprocess_layer(inputs)\n        # Add Batch Dimension\n        x = tf.expand_dims(x, axis=0)\n        non_empty_frame_idxs = tf.expand_dims(non_empty_frame_idxs, axis=0)\n        # Make Prediction\n        outputs = self.model({ 'frames': x, 'non_empty_frame_idxs': non_empty_frame_idxs })\n        # Squeeze Output 1x250 -> 250\n        outputs = tf.squeeze(outputs, axis=0)\n\n        # Return a dictionary with the output tensor\n        return {'outputs': outputs}\n\n# Define TF Lite Model\ntflite_keras_model = TFLiteModel(model)\n\n# Sanity Check\ndemo_raw_data = load_relevant_data_subset(train['file_path'].values[5])\nprint(f'demo_raw_data shape: {demo_raw_data.shape}, dtype: {demo_raw_data.dtype}')\ndemo_output = tflite_keras_model(demo_raw_data)[\"outputs\"]\nprint(f'demo_output shape: {demo_output.shape}, dtype: {demo_output.dtype}')\ndemo_prediction = demo_output.numpy().argmax()\nprint(f'demo_prediction: {demo_prediction}, correct: {train.iloc[0][\"sign_ord\"]}')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T22:18:31.963700Z","iopub.execute_input":"2023-04-26T22:18:31.964834Z","iopub.status.idle":"2023-04-26T22:18:34.368021Z","shell.execute_reply.started":"2023-04-26T22:18:31.964762Z","shell.execute_reply":"2023-04-26T22:18:34.366731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Model Converter\nkeras_model_converter = tf.lite.TFLiteConverter.from_keras_model(tflite_keras_model)\n# Convert Model\ntflite_model = keras_model_converter.convert()\n# Write Model\nwith open('/kaggle/working/model.tflite', 'wb') as f:\n    f.write(tflite_model)\n    \n# Zip Model\n!zip submission.zip /kaggle/working/model.tflite","metadata":{"execution":{"iopub.status.busy":"2023-04-26T22:18:45.802924Z","iopub.execute_input":"2023-04-26T22:18:45.803918Z","iopub.status.idle":"2023-04-26T22:19:27.760345Z","shell.execute_reply.started":"2023-04-26T22:18:45.803857Z","shell.execute_reply":"2023-04-26T22:19:27.758893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Verify TFLite model can be loaded and used for prediction\n!pip install tflite-runtime\nimport tflite_runtime.interpreter as tflite\n\ninterpreter = tflite.Interpreter(\"/kaggle/working/model.tflite\")\nfound_signatures = list(interpreter.get_signature_list().keys())\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")\n\noutput = prediction_fn(inputs=demo_raw_data)\nsign = output['outputs'].argmax()\n\nprint(\"PRED : \", ORD2SIGN.get(sign), f'[{sign}]')\nprint(\"TRUE : \", train.sign.values[0], f'[{train.sign_ord.values[0]}]')","metadata":{"execution":{"iopub.status.busy":"2023-04-26T22:19:27.763067Z","iopub.execute_input":"2023-04-26T22:19:27.763374Z","iopub.status.idle":"2023-04-26T22:19:40.509762Z","shell.execute_reply.started":"2023-04-26T22:19:27.763343Z","shell.execute_reply":"2023-04-26T22:19:40.508544Z"},"trusted":true},"execution_count":null,"outputs":[]}]}