{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sn\n\nfrom tensorflow import keras\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split, GroupShuffleSplit \n\nimport glob\nimport sys\nimport os\nimport math\nimport gc\nimport sys\nimport sklearn\nimport scipy\n\nprint(f'Tensorflow V{tf.__version__}')\nprint(f'Keras V{tf.keras.__version__}')\nprint(f'Python V{sys.version}')","metadata":{"execution":{"iopub.status.busy":"2023-05-15T23:35:20.682888Z","iopub.execute_input":"2023-05-15T23:35:20.683171Z","iopub.status.idle":"2023-05-15T23:35:30.219423Z","shell.execute_reply.started":"2023-05-15T23:35:20.683143Z","shell.execute_reply":"2023-05-15T23:35:30.216612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Import data analysis and machine learning libraries.\n\nnumpy: A scientific computing library that provides multidimensional array and matrix data structures, as well as a variety of mathematical functions to operate on them.\n\npandas: A data analysis and manipulation library that provides data structures and operations for numerical tables and time series data.\n\ntensorflow: An open-source machine learning platform that provides a comprehensive and flexible framework for developing and deploying various types of neural networks.\n\ntensorflow_addons: A contrib library that follows some of the established API patterns but implements new functionalities not available in core TensorFlow.\n\nmatplotlib: A plotting library that can generate publication-quality graphics in various formats and interactive environments.\n\nseaborn: A data visualization library based on matplotlib that provides a high-level interface for creating visually appealing and informative statistical graphics.\n\ntqdm: A fast and extensible Python and CLI progress bar.\n\nsklearn: A machine learning library that provides various classification, regression, clustering, \ndimensionality reduction, model selection, and preprocessing algorithms.\n\nglob: A module that provides a function for generating a list of files that match a given pattern.\n\nsys: A module that provides an interface to some variables and functions used or maintained by the interpreter.\n\nos: A module that provides a portable way of using operating system-dependent functionality.\n\nmath: A module that provides access to the mathematical functions defined by the C standard.\n\ngc: A module that provides an interface to optional garbage collector features.\n\nscipy: A scientific computing library that provides many user-friendly and efficient numerical routines, such as numerical integration, interpolation, optimization, linear algebra, and statistics.\n\nFinally, print the versions of TensorFlow, Keras, and Python.","metadata":{}},{"cell_type":"markdown","source":"# Matplotlib Parameter Configuration\n\n\n\n","metadata":{}},{"cell_type":"code","source":"Matplotlib Global Settings\nReset all parameters to their default values\nmpl.rcParams.update(mpl.rcParamsDefault)\n\nSet the font size of x-axis and y-axis tick labels to 16\nmpl.rcParams['xtick.labelsize'] = 16\nmpl.rcParams['ytick.labelsize'] = 16\n\nSet the font size of axis labels and titles to 18 and 24\nmpl.rcParams['axes.labelsize'] = 18\nmpl.rcParams['axes.titlesize'] = 24\n","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:26.835694Z","iopub.execute_input":"2023-04-20T20:48:26.836824Z","iopub.status.idle":"2023-04-20T20:48:26.843932Z","shell.execute_reply.started":"2023-04-20T20:48:26.836769Z","shell.execute_reply":"2023-04-20T20:48:26.84294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training Parameter Configuration\n","metadata":{}},{"cell_type":"code","source":"# If True, processing data from scratch\n# If False, loads preprocessed data\n# If True, process data from scratch; if False, load preprocessed data\n\nPREPROCESS_DATA = False\nTRAIN_MODEL = True\n# True: use 10% of participants as validation set\n# False: Use all data for training -> yields better leaderboard ranking\nUSE_VAL = False\n\nN_ROWS = 543 # Number of rows\nN_DIMS = 3 # Number of dimensions\nDIM_NAMES = ['x', 'y', 'z'] # Dimension names\nSEED = 42 # Random seed\nNUM_CLASSES = 250 # Total number of classes\nIS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\nVERBOSE = 1 if IS_INTERACTIVE else 2\n\nINPUT_SIZE = 64 # Input size\nBATCH_ALL_SIGNS_N = 4\nBATCH_SIZE = 512\nN_EPOCHS = 245\nLR_MAX = 0.00092 # Maximum learning rate\nN_WARMUP_EPOCHS = 0 # Number of warm-up epochs\nWD_RATIO = 0.054 # Weight decay ratio\nMASK_VAL = 4237 # Mask value","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:26.845422Z","iopub.execute_input":"2023-04-20T20:48:26.847574Z","iopub.status.idle":"2023-04-20T20:48:26.859099Z","shell.execute_reply.started":"2023-04-20T20:48:26.847536Z","shell.execute_reply":"2023-04-20T20:48:26.858094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"N_ROWS = 543 is setting the number of rows,\nN_DIMS = 3 is setting the number of dimensions,\nDIM_NAMES = ['x', 'y', 'z'] is setting the dimension names,\nSEED = 42 is setting the random seed,\nNUM_CLASSES = 250 is setting the number of classes,\nIS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive' is checking if it's an interactive environment,\nVERBOSE = 1 if IS_INTERACTIVE else 2 is setting the verbosity level,\nINPUT_SIZE = 64 is setting the input size,\nBATCH_ALL_SIGNS_N = 4 is setting the batch size,\nBATCH_SIZE = 512 is setting the batch size,\nN_EPOCHS = 200 is setting the number of training epochs,\nLR_MAX = 1e-3 is setting the maximum learning rate,\nN_WARMUP_EPOCHS = 0 is setting the number of warm-up epochs,\nWD_RATIO = 0.05 is setting the weight decay ratio,\nMASK_VAL = 4237 is setting the mask value.","metadata":{}},{"cell_type":"code","source":"# Prints Shape and Dtype For List Of Variables\ndef print_shape_dtype(l, names):\n    for e, n in zip(l, names):\n        print(f'{n} shape: {e.shape}, dtype: {e.dtype}')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:26.861668Z","iopub.execute_input":"2023-04-20T20:48:26.862246Z","iopub.status.idle":"2023-04-20T20:48:26.86978Z","shell.execute_reply.started":"2023-04-20T20:48:26.862209Z","shell.execute_reply":"2023-04-20T20:48:26.868769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Iterate through the list of variables and the list of names, then print the shape and data type of each variable","metadata":{}},{"cell_type":"markdown","source":"# read training data","metadata":{}},{"cell_type":"code","source":"# Read Training Data\n\nif IS_INTERACTIVE or not PREPROCESS_DATA:\n    train = pd.read_csv('/kaggle/input/asl-signs/train.csv').sample(int(5e3), random_state=SEED)\nelse:\n    train = pd.read_csv('/kaggle/input/asl-signs/train.csv')\n\nN_SAMPLES = len(train)\nprint(f'N_SAMPLES: {N_SAMPLES}')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:26.870978Z","iopub.execute_input":"2023-04-20T20:48:26.873122Z","iopub.status.idle":"2023-04-20T20:48:27.058201Z","shell.execute_reply.started":"2023-04-20T20:48:26.873078Z","shell.execute_reply":"2023-04-20T20:48:27.057064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If IS_INTERACTIVE or PREPROCESS_DATA is False,\n\nThen use the Pandas sample() function to read a random sample of 5000 rows from the CSV file.\n\nOtherwise, it reads all lines in the CSV file.","metadata":{}},{"cell_type":"markdown","source":"# add file path","metadata":{}},{"cell_type":"markdown","source":"This code defines a function called get_file_path() that takes one argument path and returns a full file path. This function is used to convert a file path to a file path in the Kaggle dataset.\nThe code applies the get_file_path() function to each element in the Pandas Series object named train['path'] using Pandas' apply() function. The code then stores the result in a new column named train['file_path'].","metadata":{}},{"cell_type":"code","source":"# Get complete file path to file\ndef get_file_path(path):\n    return f'/kaggle/input/asl-signs/{path}'\n\ntrain['file_path'] = train['path'].apply(get_file_path)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:27.059851Z","iopub.execute_input":"2023-04-20T20:48:27.060627Z","iopub.status.idle":"2023-04-20T20:48:27.073232Z","shell.execute_reply.started":"2023-04-20T20:48:27.06057Z","shell.execute_reply":"2023-04-20T20:48:27.072228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Digitally encode symbols","metadata":{}},{"cell_type":"code","source":"# Add ordinally Encoded Sign (assign number to each sign name)\n# Add ordinal encoding symbols (assigning numbers to each symbol name)\ntrain['sign_ord'] = train['sign'].astype('category').cat.codes\n\n# Dictionaries to translate sign <-> ordinal encoded sign\n# A dictionary used for conversion between symbols <-> ordinal codes\nSIGN2ORD = train[['sign', 'sign_ord']].set_index('sign').squeeze().to_dict()\nORD2SIGN = train[['sign_ord', 'sign']].set_index('sign_ord').squeeze().to_dict()","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:27.075449Z","iopub.execute_input":"2023-04-20T20:48:27.075971Z","iopub.status.idle":"2023-04-20T20:48:27.098792Z","shell.execute_reply.started":"2023-04-20T20:48:27.075924Z","shell.execute_reply":"2023-04-20T20:48:27.097919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train.head(30))\ndisplay(train.info())","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:27.100434Z","iopub.execute_input":"2023-04-20T20:48:27.100798Z","iopub.status.idle":"2023-04-20T20:48:27.134199Z","shell.execute_reply.started":"2023-04-20T20:48:27.100759Z","shell.execute_reply":"2023-04-20T20:48:27.133132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Statistical video data","metadata":{}},{"cell_type":"markdown","source":"This code is used to calculate the statistics of the number of different frames, the number of missing frames and the maximum number of frames for each video in the video dataset.\nAmong them, `N` is the value set according to the condition, `N_UNIQUE_FRAMES`, `N_MISSING_FRAMES` and `MAX_FRAME` are the arrays storing the number of different frames, the number of missing frames and the maximum number of frames respectively.\nIn the loop, the code iterates through each video in the dataset, reading the video data and counting the number of distinct frames, missing frames, and maximum frames.\nFinally, the code displays these statistics and draws a corresponding histogram.","metadata":{}},{"cell_type":"code","source":"N = int(1e3) if (IS_INTERACTIVE or not PREPROCESS_DATA) else int(10e3)  # set the value of N according to the condition\nN_UNIQUE_FRAMES = np.zeros(N, dtype=np.uint16) # Initialize the N_UNIQUE_FRAMES array\nN_MISSING_FRAMES = np.zeros(N, dtype=np.uint16) # Initialize the N_MISSING_FRAMES array\nMAX_FRAME = np.zeros(N, dtype=np.uint16) # Initialize the MAX_FRAME array\n\nPERCENTILES = [0.01, 0.05, 0.25, 0.50, 0.75, 0.95, 0.99, 0.999] # Define the PERCENTILES list\n\nfor idx, file_path in enumerate(tqdm(train['file_path'].sample(N, random_state=SEED))): # Iterate through the file paths in train['file_path']\n    df = pd.read_parquet(file_path) # Read the file\n    N_UNIQUE_FRAMES[idx] = df['frame'].nunique() # Calculate the number of unique frames in each file\n    N_MISSING_FRAMES[idx] = (df['frame'].max() - df['frame'].min()) - df['frame'].nunique() + 1 # Calculate the number of missing frames in each file\n    MAX_FRAME[idx] = df['frame'].max() # Calculate the maximum frame number in each file","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:27.13615Z","iopub.execute_input":"2023-04-20T20:48:27.136517Z","iopub.status.idle":"2023-04-20T20:48:50.259095Z","shell.execute_reply.started":"2023-04-20T20:48:27.13648Z","shell.execute_reply":"2023-04-20T20:48:50.25796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of unique frames in each video\ndisplay(pd.Series(N_UNIQUE_FRAMES).describe(percentiles=PERCENTILES).to_frame('N_UNIQUE_FRAMES'))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Unique Frames', size=24)\npd.Series(N_UNIQUE_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nxlim = math.ceil(plt.xlim()[1])\nplt.xlim(0, xlim)\nplt.xticks(np.arange(0, xlim+25, 25))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:50.263433Z","iopub.execute_input":"2023-04-20T20:48:50.263725Z","iopub.status.idle":"2023-04-20T20:48:50.889682Z","shell.execute_reply.started":"2023-04-20T20:48:50.263697Z","shell.execute_reply":"2023-04-20T20:48:50.88864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of missing frames, consecutive frames with missing intermediate frame, i.e. 1,2,4,5 -> 3 is missing\n#The number of lost frames, the consecutive frames lost in the middle frame\ndisplay(pd.Series(N_MISSING_FRAMES).describe(percentiles=PERCENTILES).to_frame('N_MISSING_FRAMES'))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Missing Frames', size=24)\npd.Series(N_MISSING_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nplt.xlim(0, math.ceil(plt.xlim()[1]))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:50.894034Z","iopub.execute_input":"2023-04-20T20:48:50.896736Z","iopub.status.idle":"2023-04-20T20:48:51.493104Z","shell.execute_reply.started":"2023-04-20T20:48:50.896695Z","shell.execute_reply":"2023-04-20T20:48:51.492053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Maximum frame number\ndisplay(pd.Series(MAX_FRAME).describe(percentiles=PERCENTILES).to_frame('MAX_FRAME'))\n\nplt.figure(figsize=(15,8))\nplt.title('Maximum Frames Index', size=24)\npd.Series(MAX_FRAME).plot(kind='hist', bins=128)\nplt.grid()\nplt.xlim(0, math.ceil(plt.xlim()[1]))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:51.497535Z","iopub.execute_input":"2023-04-20T20:48:51.499899Z","iopub.status.idle":"2023-04-20T20:48:52.10409Z","shell.execute_reply.started":"2023-04-20T20:48:51.499844Z","shell.execute_reply":"2023-04-20T20:48:52.103035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Landmark Indices key point index","metadata":{}},{"cell_type":"markdown","source":"In machine learning, Landmark Indices usually refers to the index of face key points. Face key points are some specific points on the face, such as eyes, nose, mouth, etc., which can be used for tasks such as face recognition and expression recognition. In machine learning, we can use these key points to train models to achieve tasks such as face recognition. ¹","metadata":{}},{"cell_type":"code","source":"#Define three data types: left_hand, pose, and right_hand\nUSE_TYPES = ['left_hand', 'pose', 'right_hand']\n\n#Define the starting index in the original data\nSTART_IDX = 468\n\n#Define the lip landmark indices in the original data (total 40)\nLIPS_IDXS0 = np.array([\n61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n])\n\n#Define the left hand landmark indices in the original data (total 21)\nLEFT_HAND_IDXS0 = np.arange(468, 489)\n\n#Define the right hand landmark indices in the original data (total 21)\nRIGHT_HAND_IDXS0 = np.arange(522, 543)\n\n#Define the left pose landmark indices in the original data (total 5)\nLEFT_POSE_IDXS0 = np.array([502, 504, 506, 508, 510])\n\n#Define the right pose landmark indices in the original data (total 5)\nRIGHT_POSE_IDXS0 = np.array([503, 505, 507, 509, 511])\n\n#Define the left-dominant landmark indices, including lips, left hand, and left pose (total 66)\nLANDMARK_IDXS_LEFT_DOMINANT0 = np.concatenate((LIPS_IDXS0, LEFT_HAND_IDXS0, LEFT_POSE_IDXS0))\n\n#Define the right-dominant landmark indices, including lips, right hand, and right pose (total 66)\nLANDMARK_IDXS_RIGHT_DOMINANT0 = np.concatenate((LIPS_IDXS0, RIGHT_HAND_IDXS0, RIGHT_POSE_IDXS0))\n\n#Define all hand landmark indices, including left hand and right hand (total 42)\nHAND_IDXS0 = np.concatenate((LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0), axis=0)\n\n#Define the number of columns in the processed data (equals to 66)\nN_COLS = LANDMARK_IDXS_LEFT_DOMINANT0.size\n\n#Define the lip landmark indices in the processed data (from 0 to 39)\nLIPS_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LIPS_IDXS0)).squeeze()\n\n#Define the left hand landmark indices in the processed data (from 40 to 60)\nLEFT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_HAND_IDXS0)).squeeze()\n\n#Define the right hand landmark indices in the processed data (from 40 to 60)\nRIGHT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, RIGHT_HAND_IDXS0)).squeeze()\n\n#Define all hand landmark indices in the processed data (from 40 to 81)\nHAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, HAND_IDXS0)).squeeze()\n\n#Define the pose landmark indices in the processed data (from 61 to 65)\nPOSE_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_POSE_IDXS0)).squeeze()\n\n#Print the length of hand landmark indices","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:52.108633Z","iopub.execute_input":"2023-04-20T20:48:52.111008Z","iopub.status.idle":"2023-04-20T20:48:52.131705Z","shell.execute_reply.started":"2023-04-20T20:48:52.110967Z","shell.execute_reply":"2023-04-20T20:48:52.130372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Define the starting position of lip landmarks, which is 0\nLIPS_START = 0\n\n#Define the starting position of left hand landmarks, which is the number of lip landmarks\nLEFT_HAND_START = LIPS_IDXS.size\n\n#Define the starting position of right hand landmarks, which is the starting position of left hand landmarks plus the number of left hand landmarks\nRIGHT_HAND_START = LEFT_HAND_START + LEFT_HAND_IDXS.size\n\n#Define the starting position of pose landmarks, which is the starting position of right hand landmarks plus the number of right hand landmarks\nPOSE_START = RIGHT_HAND_START + RIGHT_HAND_IDXS.size\n\n#Print the starting positions of different landmarks\nprint(f'LIPS_START: {LIPS_START}, LEFT_HAND_START: {LEFT_HAND_START}, RIGHT_HAND_START: {RIGHT_HAND_START}, POSE_START: {POSE_START}')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:52.133627Z","iopub.execute_input":"2023-04-20T20:48:52.13408Z","iopub.status.idle":"2023-04-20T20:48:52.147918Z","shell.execute_reply.started":"2023-04-20T20:48:52.134043Z","shell.execute_reply":"2023-04-20T20:48:52.14671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process Data Tensorflow","metadata":{}},{"cell_type":"code","source":"# Source: https://www.kaggle.com/competitions/asl-signs/overview/evaluation\nROWS_PER_FRAME = 543 # number of landmarks per frame\n\n#load_relevant_data_subset load relevant data subset\n#Extract the three columns of data x, y, and z from the data set, divide the data set according to the number of landmarks in each frame, and return a three-dimensional array.\n#The first dimension represents the number of frames, the second dimension represents the number of landmarks per frame, and the third dimension represents the three coordinate axes of x, y, and z.\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:52.153045Z","iopub.execute_input":"2023-04-20T20:48:52.154309Z","iopub.status.idle":"2023-04-20T20:48:52.163366Z","shell.execute_reply.started":"2023-04-20T20:48:52.154243Z","shell.execute_reply":"2023-04-20T20:48:52.162352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code defines a constant called ROWS_PER_FRAME with a value of 543, representing the number of landmarks per frame. The function load_relevant_data_subset(pq_path) reads the parquet file under the specified path, extracts the three columns of data x, y, and z in the data set, divides the data set according to the number of landmarks in each frame, and returns a three-dimensional array. The first dimension represents the number of frames, the second dimension represents the number of landmarks per frame, and the third dimension represents the three coordinate axes of x, y, and z. The return value type of the function is numpy.ndarray, and the data type is np.float32.","metadata":{}},{"cell_type":"markdown","source":"# Customize a data preprocessing layer using TF","metadata":{}},{"cell_type":"markdown","source":"The PreprocessLayer class is a custom layer inherited from TensorFlow's tf.keras.layers.Layer class for processing data in TensorFlow Lite models. This custom layer contains functions for processing input data.","metadata":{}},{"cell_type":"markdown","source":"First, in the __init__ method, the layer creates a constant named normalisation_correction. This constant is a matrix with the number of rows equal to the number of landmarks of a specific type in the data, and 3 columns for the x, y, and z coordinates. This matrix is used to correct the camera's shooting direction, swapping left and right hands.\n\nThe layer also defines a method called pad_edge, which is used to pad a given tensor with a certain number of repeated elements on the left or right side.\n\nNext, the layer decorates a call method with the @tf.function decorator to handle input data.\n\nThe method first calculates the number of frames (N_FRAMES0) in the input data. It then finds the dominant hand landmarks in the data by computing the sum of coordinates for left and right hands, and determines the frames to keep by counting the non-NaN values in the dominant hand for each frame. It collects the landmark data from the input data using these indices.\n\nThe method then converts the data type of the frame indices from integers to floats and normalizes them to start from 0. It calculates the number of frames (N_FRAMES) again from the filtered data and collects the landmarks of a specific type from these data. If the number of frames in the data is less than the specified input size (INPUT_SIZE), it pads with -1 to extend the number of frames to the specified input size and replaces NaN values with 0. If the number of frames is greater than the specified input size, it reduces it to the specified input size using repeated data and fills any missing data.\n\nFinally, the method returns the processed data and the corresponding frame indices.","metadata":{}},{"cell_type":"markdown","source":"his code snippet is a TensorFlow Keras custom layer named PreprocessLayer. It is used for preprocessing input data, including handling, padding, and normalization of hand landmark data.\n\nThe main functionalities of this layer include:\n\nInitialization: In the __init__ method, the layer initializes itself by calling the __init__ method of the parent class tf.keras.layers.Layer. It also defines a constant tensor named normalisation_correction and stores its transpose in self.normalisation_correction.\n\npad_edge method: This method is used to pad data at the edges of the input data based on the specified padding direction ('LEFT' or 'RIGHT') and the number of repetitions.\n\ncall method: This method defines the operations of a computational graph using the @tf.function decorator to preprocess the input data. The steps involved are as follows:\n\na. Obtaining the first dimension (number of frames) of the input data and storing it in the variable N_FRAMES0.\n\nb. Determining the dominant hand (left or right) and calculating the sum of non-NaN values for hand landmarks in each frame, storing the results in left_hand_sum and right_hand_sum.\n\nc. Based on the dominant hand, calculating the sum of non-NaN values for the dominant hand landmarks in each frame and storing the results in frames_hands_non_nan_sum.\n\nd. Using the results in frames_hands_non_nan_sum, finding the indices of frames with non-empty hand landmarks and storing them in non_empty_frames_idxs.\n\ne. Filtering the input data based on non_empty_frames_idxs and performing a series of normalization and padding operations. Finally, returning the processed data and the padded frame indices.\n\nIn summary, the PreprocessLayer custom layer is primarily used for preprocessing input data, including handling, padding, and normalization of hand landmark data, to meet the requirements of subsequent model inputs.","metadata":{}},{"cell_type":"code","source":"\"\"\"\n    Tensorflow layer to process data in TFLite\n    Data needs to be processed in the model itself, so we can not use Python\n\"\"\" \nclass PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__()\n        normalisation_correction = tf.constant([\n                    # Add 0.50 to left hand (original right hand) and substract 0.50 of right hand (original left hand)\n                    [0] * len(LIPS_IDXS) + [0.50] * len(LEFT_HAND_IDXS) + [0.50] * len(POSE_IDXS),\n                    # Y coordinates stay intact\n                    [0] * len(LANDMARK_IDXS_LEFT_DOMINANT0),\n                    # Z coordinates stay intact\n                    [0] * len(LANDMARK_IDXS_LEFT_DOMINANT0),\n                ],\n                dtype=tf.float32,\n            )\n        self.normalisation_correction = tf.transpose(normalisation_correction, [1,0])\n        \n    def pad_edge(self, t, repeats, side):\n        if side == 'LEFT':\n            return tf.concat((tf.repeat(t[:1], repeats=repeats, axis=0), t), axis=0)\n        elif side == 'RIGHT':\n            return tf.concat((t, tf.repeat(t[-1:], repeats=repeats, axis=0)), axis=0)\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,N_ROWS,N_DIMS], dtype=tf.float32),),\n    )\n    def call(self, data0):\n        # Number of Frames in Video\n        N_FRAMES0 = tf.shape(data0)[0]\n        \n        # Find dominant hand by comparing summed absolute coordinates\n        left_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0, 1))\n        right_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0, 1))\n        left_dominant = left_hand_sum >= right_hand_sum\n        \n        # Count non NaN Hand values in each frame for the dominant hand\n        if left_dominant:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0, 1),\n                    axis=[1, 2],\n                )\n        else:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0, 1),\n                    axis=[1, 2],\n                )\n        \n        # Find frames indices with coordinates of dominant hand\n        non_empty_frames_idxs = tf.where(frames_hands_non_nan_sum > 0)\n        non_empty_frames_idxs = tf.squeeze(non_empty_frames_idxs, axis=1)\n        # Filter frames\n        data = tf.gather(data0, non_empty_frames_idxs, axis=0)\n        \n        # Cast Indices in float32 to be compatible with Tensorflow Lite\n        non_empty_frames_idxs = tf.cast(non_empty_frames_idxs, tf.float32)\n        # Normalize to start with 0\n        non_empty_frames_idxs -= tf.reduce_min(non_empty_frames_idxs)\n        \n        # Number of Frames in Filtered Video\n        N_FRAMES = tf.shape(data)[0]\n        \n        # Gather Relevant Landmark Columns\n        if left_dominant:\n            data = tf.gather(data, LANDMARK_IDXS_LEFT_DOMINANT0, axis=1)\n        else:\n            data = tf.gather(data, LANDMARK_IDXS_RIGHT_DOMINANT0, axis=1)\n            data = (\n                    self.normalisation_correction + (\n                        (data - self.normalisation_correction) * tf.where(self.normalisation_correction != 0, -1.0, 1.0))\n                )\n        \n        # Video fits in INPUT_SIZE\n        if N_FRAMES < INPUT_SIZE:\n            # Pad With -1 to indicate padding\n            non_empty_frames_idxs = tf.pad(non_empty_frames_idxs, [[0, INPUT_SIZE-N_FRAMES]], constant_values=-1)\n            # Pad Data With Zeros\n            data = tf.pad(data, [[0, INPUT_SIZE-N_FRAMES], [0,0], [0,0]], constant_values=0)\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            return data, non_empty_frames_idxs\n        # Video needs to be downsampled to INPUT_SIZE\n        else:\n            # Repeat\n            if N_FRAMES < INPUT_SIZE**2:\n                repeats = tf.math.floordiv(INPUT_SIZE * INPUT_SIZE, N_FRAMES0)\n                data = tf.repeat(data, repeats=repeats, axis=0)\n                non_empty_frames_idxs = tf.repeat(non_empty_frames_idxs, repeats=repeats, axis=0)\n\n            # Pad To Multiple Of Input Size\n            pool_size = tf.math.floordiv(len(data), INPUT_SIZE)\n            if tf.math.mod(len(data), INPUT_SIZE) > 0:\n                pool_size += 1\n\n            if pool_size == 1:\n                pad_size = (pool_size * INPUT_SIZE) - len(data)\n            else:\n                pad_size = (pool_size * INPUT_SIZE) % len(data)\n\n            # Pad Start/End with Start/End value\n            pad_left = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(INPUT_SIZE, 2)\n            pad_right = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(INPUT_SIZE, 2)\n            if tf.math.mod(pad_size, 2) > 0:\n                pad_right += 1\n\n            # Pad By Concatenating Left/Right Edge Values\n            data = self.pad_edge(data, pad_left, 'LEFT')\n            data = self.pad_edge(data, pad_right, 'RIGHT')\n\n            # Pad Non Empty Frame Indices\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_left, 'LEFT')\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_right, 'RIGHT')\n\n            # Reshape to Mean Pool\n            data = tf.reshape(data, [INPUT_SIZE, -1, N_COLS, N_DIMS])\n            non_empty_frames_idxs = tf.reshape(non_empty_frames_idxs, [INPUT_SIZE, -1])\n\n            # Mean Pool\n            data = tf.experimental.numpy.nanmean(data, axis=1)\n            non_empty_frames_idxs = tf.experimental.numpy.nanmean(non_empty_frames_idxs, axis=1)\n\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            \n            return data, non_empty_frames_idxs\n    \npreprocess_layer = PreprocessLayer()","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:52.167839Z","iopub.execute_input":"2023-04-20T20:48:52.170496Z","iopub.status.idle":"2023-04-20T20:48:54.842283Z","shell.execute_reply.started":"2023-04-20T20:48:52.170453Z","shell.execute_reply":"2023-04-20T20:48:54.841228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Interpolate NaN Values","metadata":{}},{"cell_type":"markdown","source":"Interpolate NaN Values ​​refers to the use of interpolation to fill the NaN values ​​​​in the data. NaN is a class of values ​​of numeric data types in computer science that represent undefined or unrepresentable values. NaN is the abbreviation of Not a Number, understood as not a value. In computers, NaN is usually used to represent invalid or undefined operation results, such as 0/0, ∞-∞, etc. 1.","metadata":{}},{"cell_type":"code","source":"\"\"\"\n    face: 0:468\n    left_hand: 468:489\n    pose: 489:522\n    right_hand: 522:544\n    get data from file_path\n    The first line of code calls the load_relevant_data_subset function to load the original data from the file path.\n    The second line of code calls the preprocess_layer function, which uses Tensorflow to process the data.\n    Finally return the processed data.\n\"\"\"\ndef get_data(file_path):\n    # Load Raw Data\n    data = load_relevant_data_subset(file_path)\n    # Process Data Using Tensorflow\n    data = preprocess_layer(data)\n    \n    return data","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:54.844024Z","iopub.execute_input":"2023-04-20T20:48:54.844403Z","iopub.status.idle":"2023-04-20T20:48:54.851163Z","shell.execute_reply.started":"2023-04-20T20:48:54.844361Z","shell.execute_reply":"2023-04-20T20:48:54.850163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Dataset","metadata":{}},{"cell_type":"code","source":"# Get the full dataset\ndef preprocess_data():\n    # Create arrays to save data\n    X = np.zeros([N_SAMPLES, INPUT_SIZE, N_COLS, N_DIMS], dtype=np.float32)\n    y = np.zeros([N_SAMPLES], dtype=np.int32)\n    NON_EMPTY_FRAME_IDXS = np.full([N_SAMPLES, INPUT_SIZE], -1, dtype=np.float32)\n\n    # Fill X/y\n    for row_idx, (file_path, sign_ord) in enumerate(tqdm(train[['file_path', 'sign_ord']].values)):\n        # Log message every 5000 samples\n        if row_idx % 5000 == 0:\n            print(f'Generated {row_idx}/{N_SAMPLES}')\n\n        data, non_empty_frame_idxs = get_data(file_path)\n        X[row_idx] = data\n        y[row_idx] = sign_ord\n        NON_EMPTY_FRAME_IDXS[row_idx] = non_empty_frame_idxs\n        # Sanity check, data should not contain NaN values\n        if np.isnan(data).sum() > 0:\n            print(row_idx)\n            return data\n\n    # Save X/y\n    np.save('X.npy', X)\n    np.save('y.npy', y)\n    np.save('NON_EMPTY_FRAME_IDXS.npy', NON_EMPTY_FRAME_IDXS)\n    \n    # Save Validation\n    splitter = GroupShuffleSplit(test_size=0.10, n_splits=2, random_state=SEED)\n    PARTICIPANT_IDS = train['participant_id'].values\n    train_idxs, val_idxs = next(splitter.split(X, y, groups=PARTICIPANT_IDS))\n\n    # Save Train\n    X_train = X[train_idxs]\n    NON_EMPTY_FRAME_IDXS_TRAIN = NON_EMPTY_FRAME_IDXS[train_idxs]\n    y_train = y[train_idxs]\n    np.save('X_train.npy', X_train)\n    np.save('y_train.npy', y_train)\n    np.save('NON_EMPTY_FRAME_IDXS_TRAIN.npy', NON_EMPTY_FRAME_IDXS_TRAIN)\n    # Save Validation\n    X_val = X[val_idxs]\n    NON_EMPTY_FRAME_IDXS_VAL = NON_EMPTY_FRAME_IDXS[val_idxs]\n    y_val = y[val_idxs]\n    np.save('X_val.npy', X_val)\n    np.save('y_val.npy', y_val)\n    np.save('NON_EMPTY_FRAME_IDXS_VAL.npy', NON_EMPTY_FRAME_IDXS_VAL)\n    # Split Statistics\n    print(f'Patient ID Intersection Train/Val: {set(PARTICIPANT_IDS[train_idxs]).intersection(PARTICIPANT_IDS[val_idxs])}')\n    print(f'X_train shape: {X_train.shape}, X_val shape: {X_val.shape}')\n    print(f'y_train shape: {y_train.shape}, y_val shape: {y_val.shape}')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:54.852707Z","iopub.execute_input":"2023-04-20T20:48:54.853388Z","iopub.status.idle":"2023-04-20T20:48:54.872678Z","shell.execute_reply.started":"2023-04-20T20:48:54.85335Z","shell.execute_reply":"2023-04-20T20:48:54.871437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess All Data From Scratch\nif PREPROCESS_DATA:\n    preprocess_data()\n    ROOT_DIR = '.'\nelse:\n    ROOT_DIR = '/kaggle/input/gislr-dataset-public'\n    \n# Load Data\nif USE_VAL:\n    # Load Train\n    X_train = np.load(f'{ROOT_DIR}/X_train.npy')\n    y_train = np.load(f'{ROOT_DIR}/y_train.npy')\n    NON_EMPTY_FRAME_IDXS_TRAIN = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS_TRAIN.npy')\n    # Load Val\n    X_val = np.load(f'{ROOT_DIR}/X_val.npy')\n    y_val = np.load(f'{ROOT_DIR}/y_val.npy')\n    NON_EMPTY_FRAME_IDXS_VAL = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS_VAL.npy')\n    # Define validation Data\n    validation_data = ({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, y_val)\nelse:\n    X_train = np.load(f'{ROOT_DIR}/X.npy')\n    y_train = np.load(f'{ROOT_DIR}/y.npy')\n    NON_EMPTY_FRAME_IDXS_TRAIN = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS.npy')\n    validation_data = None\n\n# Train \nprint_shape_dtype([X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN], ['X_train', 'y_train', 'NON_EMPTY_FRAME_IDXS_TRAIN'])\n# Val\nif USE_VAL:\n    print_shape_dtype([X_val, y_val, NON_EMPTY_FRAME_IDXS_VAL], ['X_val', 'y_val', 'NON_EMPTY_FRAME_IDXS_VAL'])\n# Sanity Check\nprint(f'# NaN Values X_train: {np.isnan(X_train).sum()}')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:48:54.87419Z","iopub.execute_input":"2023-04-20T20:48:54.874721Z","iopub.status.idle":"2023-04-20T20:49:29.582066Z","shell.execute_reply.started":"2023-04-20T20:48:54.874682Z","shell.execute_reply":"2023-04-20T20:49:29.580861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Class Count\ndisplay(pd.Series(y_train).value_counts().to_frame('Class Count').iloc[[0,1,2,3,4, -5,-4,-3,-2,-1]])","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:49:29.58381Z","iopub.execute_input":"2023-04-20T20:49:29.584216Z","iopub.status.idle":"2023-04-20T20:49:29.599558Z","shell.execute_reply.started":"2023-04-20T20:49:29.584176Z","shell.execute_reply":"2023-04-20T20:49:29.597868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Number Of Frames","metadata":{}},{"cell_type":"markdown","source":"Generate a Waterfall Plot of the number of non-empty frames for samples in the dataset","metadata":{}},{"cell_type":"code","source":"# Vast majority of samples fits has less than 32 non empty frames\nN_EMPTY_FRAMES = (NON_EMPTY_FRAME_IDXS_TRAIN != -1).sum(axis=1) \nN_EMPTY_FRAMES_WATERFALL = []\nfor n in tqdm(range(1,INPUT_SIZE+1)):\n    N_EMPTY_FRAMES_WATERFALL.append(sum(N_EMPTY_FRAMES >= n) / len(NON_EMPTY_FRAME_IDXS_TRAIN) * 100)\n\nplt.figure(figsize=(18,10))\nplt.title('Waterfall Plot For Number Of Non Empty Frames')\npd.Series(N_EMPTY_FRAMES_WATERFALL).plot(kind='bar')\nplt.grid(axis='y')\nplt.xticks(np.arange(INPUT_SIZE), np.arange(1, INPUT_SIZE+1))\nplt.xlabel('Number of Non Empty Frames', size=16)\nplt.yticks(np.arange(0, 100+10, 10))\nplt.ylim(0, 100)\nplt.ylabel('Percentage of Samples With At Least N Non Empty Frames', size=16)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:49:29.600992Z","iopub.execute_input":"2023-04-20T20:49:29.601927Z","iopub.status.idle":"2023-04-20T20:49:43.077901Z","shell.execute_reply.started":"2023-04-20T20:49:29.601871Z","shell.execute_reply":"2023-04-20T20:49:43.076907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Percentage of Frames Filled","metadata":{}},{"cell_type":"code","source":"# Percentage of frames filled, this is the maximum fill percentage of each landmark\nP_DATA_FILLED = (NON_EMPTY_FRAME_IDXS_TRAIN != -1).sum() / NON_EMPTY_FRAME_IDXS_TRAIN.size * 100\nprint(f'P_DATA_FILLED: {P_DATA_FILLED:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:49:43.07897Z","iopub.execute_input":"2023-04-20T20:49:43.079295Z","iopub.status.idle":"2023-04-20T20:49:43.095822Z","shell.execute_reply.started":"2023-04-20T20:49:43.079253Z","shell.execute_reply":"2023-04-20T20:49:43.094542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Statistics - Lips","metadata":{}},{"cell_type":"code","source":"# Percentage of Lips Measurements\nP_LEFT_LIPS_MEASUREMENTS = (X_train[:,:,LIPS_IDXS] != 0).sum() / X_train[:,:,LIPS_IDXS].size / P_DATA_FILLED * 1e4\nprint(f'P_LEFT_LIPS_MEASUREMENTS: {P_LEFT_LIPS_MEASUREMENTS:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:49:43.097845Z","iopub.execute_input":"2023-04-20T20:49:43.098234Z","iopub.status.idle":"2023-04-20T20:50:00.932541Z","shell.execute_reply.started":"2023-04-20T20:49:43.098197Z","shell.execute_reply":"2023-04-20T20:50:00.931216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"What this code does is calculate the mean and standard deviation of the lips. It uses numpy and matplotlib libraries. Among them, the np.transpose function converts the dimension of the X_train array from (number of samples, number of time steps, number of features) to (number of features, number of time steps, number of samples), and then the reshape function converts it into (number of features, number of time steps number * number of samples) of shape. Finally, for each feature and each sample, compute the mean and standard deviation of the nonzero elements and store the results in the LIPS_MEAN_X, LIPS_MEAN_Y, LIPS_STD_X, and LIPS_STD_Y arrays. These arrays are finally combined into LIPS_MEAN and LIPS_STD arrays and returned as the output of the function.","metadata":{}},{"cell_type":"code","source":"def get_lips_mean_std():\n    # LIPS\n    LIPS_MEAN_X = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_MEAN_Y = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_STD_X = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_STD_Y = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n\n    fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,LIPS_IDXS], [2,3,0,1]).reshape([LIPS_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                LIPS_MEAN_X[col] = v.mean()\n                LIPS_STD_X[col] = v.std()\n            if dim == 1: # Y\n                LIPS_MEAN_Y[col] = v.mean()\n                LIPS_STD_Y[col] = v.std()\n\n            axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    for ax, dim_name in zip(axes, DIM_NAMES):\n        ax.set_title(f'Lips {dim_name.upper()} Dimension', size=24)\n        ax.tick_params(axis='x', labelsize=8)\n        ax.grid(axis='y')\n\n    plt.subplots_adjust(hspace=0.50)\n    plt.show()\n\n    LIPS_MEAN = np.array([LIPS_MEAN_X, LIPS_MEAN_Y]).T\n    LIPS_STD = np.array([LIPS_STD_X, LIPS_STD_Y]).T\n    \n    return LIPS_MEAN, LIPS_STD\n\nLIPS_MEAN, LIPS_STD = get_lips_mean_std()","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:00.934291Z","iopub.execute_input":"2023-04-20T20:50:00.935074Z","iopub.status.idle":"2023-04-20T20:50:20.089219Z","shell.execute_reply.started":"2023-04-20T20:50:00.935028Z","shell.execute_reply":"2023-04-20T20:50:20.08808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Statistics - Hands","metadata":{}},{"cell_type":"code","source":"# Verify Normalised to Left Hand Dominant\nP_LEFT_HAND_MEASUREMENTS = (X_train[:,:,LEFT_HAND_IDXS] != 0).sum() / X_train[:,:,LEFT_HAND_IDXS].size / P_DATA_FILLED * 1e4\n# P_RIGHT_HAND_MEASUREMENTS = (X_train[:,:,RIGHT_HAND_IDXS] != 0).sum() / X_train[:,:,RIGHT_HAND_IDXS].size / P_DATA_FILLED * 1e4\nprint(f'P_LEFT_HAND_MEASUREMENTS: {P_LEFT_HAND_MEASUREMENTS:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:20.090698Z","iopub.execute_input":"2023-04-20T20:50:20.09119Z","iopub.status.idle":"2023-04-20T20:50:29.395979Z","shell.execute_reply.started":"2023-04-20T20:50:20.091148Z","shell.execute_reply":"2023-04-20T20:50:29.394701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_left_right_hand_mean_std():\n    # LEFT HAND\n    LEFT_HANDS_MEAN_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_MEAN_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_STD_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_STD_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n\n    fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,LEFT_HAND_IDXS], [2,3,0,1]).reshape([LEFT_HAND_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                LEFT_HANDS_MEAN_X[col] = v.mean()\n                LEFT_HANDS_STD_X[col] = v.std()\n            if dim == 1: # Y\n                LEFT_HANDS_MEAN_Y[col] = v.mean()\n                LEFT_HANDS_STD_Y[col] = v.std()\n            # Plot\n            axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    for ax, dim_name in zip(axes, DIM_NAMES):\n        ax.set_title(f'Hands {dim_name.upper()} Dimension', size=24)\n        ax.tick_params(axis='x', labelsize=8)\n        ax.grid(axis='y')\n\n    plt.subplots_adjust(hspace=0.50)\n    plt.show()\n\n    LEFT_HANDS_MEAN = np.array([LEFT_HANDS_MEAN_X, LEFT_HANDS_MEAN_Y]).T\n    LEFT_HANDS_STD = np.array([LEFT_HANDS_STD_X, LEFT_HANDS_STD_Y]).T\n    \n    return LEFT_HANDS_MEAN, LEFT_HANDS_STD\n\nLEFT_HANDS_MEAN, LEFT_HANDS_STD = get_left_right_hand_mean_std()","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:29.397728Z","iopub.execute_input":"2023-04-20T20:50:29.398463Z","iopub.status.idle":"2023-04-20T20:50:40.841827Z","shell.execute_reply.started":"2023-04-20T20:50:29.39842Z","shell.execute_reply":"2023-04-20T20:50:40.840833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature Statistics - Pose","metadata":{}},{"cell_type":"code","source":"# Percentage of Lips Measurements\nP_POSE_MEASUREMENTS = (X_train[:,:,POSE_IDXS] != 0).sum() / X_train[:,:,POSE_IDXS].size / P_DATA_FILLED * 1e4\nprint(f'P_POSE_MEASUREMENTS: {P_POSE_MEASUREMENTS:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:40.84332Z","iopub.execute_input":"2023-04-20T20:50:40.846165Z","iopub.status.idle":"2023-04-20T20:50:42.727113Z","shell.execute_reply.started":"2023-04-20T20:50:40.846123Z","shell.execute_reply":"2023-04-20T20:50:42.725857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_pose_mean_std():\n    # POSE\n    POSE_MEAN_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_MEAN_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_STD_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_STD_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\n\n    fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,POSE_IDXS], [2,3,0,1]).reshape([POSE_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                POSE_MEAN_X[col] = v.mean()\n                POSE_STD_X[col] = v.std()\n            if dim == 1: # Y\n                POSE_MEAN_Y[col] = v.mean()\n                POSE_STD_Y[col] = v.std()\n\n            axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    for ax, dim_name in zip(axes, DIM_NAMES):\n        ax.set_title(f'Pose {dim_name.upper()} Dimension', size=24)\n        ax.tick_params(axis='x', labelsize=8)\n        ax.grid(axis='y')\n\n    plt.subplots_adjust(hspace=0.50)\n    plt.show()\n\n    POSE_MEAN = np.array([POSE_MEAN_X, POSE_MEAN_Y]).T\n    POSE_STD = np.array([POSE_STD_X, POSE_STD_Y]).T\n    \n    return POSE_MEAN, POSE_STD\n\nPOSE_MEAN, POSE_STD = get_pose_mean_std()","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:42.734766Z","iopub.execute_input":"2023-04-20T20:50:42.735313Z","iopub.status.idle":"2023-04-20T20:50:45.544913Z","shell.execute_reply.started":"2023-04-20T20:50:42.73528Z","shell.execute_reply":"2023-04-20T20:50:45.543876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Samples","metadata":{}},{"cell_type":"code","source":"# Custom sampler to get a batch containing N times all signs\ndef get_train_batch_all_signs(X, y, NON_EMPTY_FRAME_IDXS, n=BATCH_ALL_SIGNS_N):\n    # Arrays to store batch in\n    X_batch = np.zeros([NUM_CLASSES*n, INPUT_SIZE, N_COLS, N_DIMS], dtype=np.float32)\n    y_batch = np.arange(0, NUM_CLASSES, step=1/n, dtype=np.float32).astype(np.int64)\n    non_empty_frame_idxs_batch = np.zeros([NUM_CLASSES*n, INPUT_SIZE], dtype=np.float32)\n    \n    # Dictionary mapping ordinally encoded sign to corresponding sample indices\n    CLASS2IDXS = {}\n    for i in range(NUM_CLASSES):\n        CLASS2IDXS[i] = np.argwhere(y == i).squeeze().astype(np.int32)\n            \n    while True:\n        # Fill batch arrays\n        for i in range(NUM_CLASSES):\n            idxs = np.random.choice(CLASS2IDXS[i], n)\n            X_batch[i*n:(i+1)*n] = X[idxs]\n            non_empty_frame_idxs_batch[i*n:(i+1)*n] = NON_EMPTY_FRAME_IDXS[idxs]\n        \n        yield { 'frames': X_batch, 'non_empty_frame_idxs': non_empty_frame_idxs_batch }, y_batch","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:45.546754Z","iopub.execute_input":"2023-04-20T20:50:45.547813Z","iopub.status.idle":"2023-04-20T20:50:45.556712Z","shell.execute_reply.started":"2023-04-20T20:50:45.547765Z","shell.execute_reply":"2023-04-20T20:50:45.555649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code defines a generator function get_train_batch_all_signs to generate a specified number (n) of training batches of all sign language signs.\n\nThis function takes as input a sign language dataset (X and y), a non-empty frame index set (NON_EMPTY_FRAME_IDXS ), and the number of all sign language tokens in a batch (n), and produces a training batch of NUM_CLASSES * n samples. This training batch contains a frames dictionary and a non_empty_frame_idxs dictionary to store the sample's sign language frames and corresponding non-empty frame indices. The y_batch array contains the serial numbers of all sign language tokens.\n\nThe main logic of the function is to loop through all sign language tokens, select n samples from each token, and add them to the batch array. The generator keeps looping through these samples so that the model can keep getting training data throughout the training process.","metadata":{}},{"cell_type":"code","source":"dummy_dataset = get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN)\nX_batch, y_batch = next(dummy_dataset)\n\nfor k, v in X_batch.items():\n    print(f'{k} shape: {v.shape}, dtype: {v.dtype}')\n\n# Batch shape/dtype\nprint(f'y_batch shape: {y_batch.shape}, dtype: {y_batch.dtype}')\n# Verify each batch contains each sign exactly N times\ndisplay(pd.Series(y_batch).value_counts().to_frame('Counts'))","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:45.558212Z","iopub.execute_input":"2023-04-20T20:50:45.558875Z","iopub.status.idle":"2023-04-20T20:50:45.641176Z","shell.execute_reply.started":"2023-04-20T20:50:45.558838Z","shell.execute_reply":"2023-04-20T20:50:45.640131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"What this code does is to randomly select BATCH_ALL_SIGNS_N samples from the X_train and y_train datasets to generate a training batch containing all sign language markers.\n\nThe generator function get_train_batch_all_signs will create an infinite loop generator that produces training batches containing all sign language tokens. For testing purposes, dummy_dataset is a batch obtained from this generator, which contains X_batch, y_batch and NON_EMPTY_FRAME_IDXS_TRAIN dictionaries, and a constant BATCH_ALL_SIGNS_N containing the number of all sign language tokens.\n\nIn the above code, X_batch is a dictionary which contains frames and non_empty_frame_idxs dictionaries whose shapes and data types are printed. Also, the shape and data type of the y_batch array are printed. Finally, verify that each sign language sign was included BATCH_ALL_SIGNS_N times using the pd.Series(y_batch).value_counts() function.","metadata":{}},{"cell_type":"markdown","source":"# Model Config","metadata":{}},{"cell_type":"markdown","source":"This code defines some constants and variables for the machine learning model.\n\n`LAYER_NORM_EPS` is a constant that sets the epsilon value for layer normalization.\n\n`LIPS_UNITS`, `HANDS_UNITS`, `POSE_UNITS` and `UNITS` are variables used to set the number of dense layer units for keypoints, final embedding and transformer embedding size.\n\n`NUM_BLOCKS` and `MLP_RATIO` are variables that set the number of transformer blocks and MLP ratios.\n\n`EMBEDDING_DROPOUT`, `MLP_DROPOUT_RATIO` and `CLASSIFIER_DROPOUT_RATIO` are variables used to set dropout ratios for embeddings, MLPs and classifiers.\n\n`INIT_HE_UNIFORM`, `INIT_GLOROT_UNIFORM` and `INIT_ZEROS` are variables, initializers for setting weights.\n\n`GELU` is a variable used to set the activation function.\n\nThe last line prints the value of `UNITS`.","metadata":{}},{"cell_type":"code","source":"# Epsilon value for layer normalisation\n#epsilon values ​​In machine learning, layer normalization is a normalization technique used to normalize the input in each layer of a neural network. This helps to speed up training and improve the accuracy of the model.\n#In layer normalization, the epsilon value is a constant used to set the epsilon value of layer normalization. It is a very small number, usually set to 1e-5 or 1e-6.\n#Its function is to prevent the denominator from being zero, so as to avoid the instability of numerical calculation.\n\n\nLAYER_NORM_EPS = 1e-6\n\n# Dense layer units for landmarks\nLIPS_UNITS = 384\nHANDS_UNITS = 384\nPOSE_UNITS = 384\n# final embedding and transformer embedding size\nUNITS = 512\n\n# Transformer\n##mlp_ratio represents the first fully connected layer ascending channel multiple；\nNUM_BLOCKS = 2\nMLP_RATIO = 4\n\n# Dropout\n#The very important nature of the model is non-linearity,\n#At the same time, for the generalization ability of the model, it is necessary to add random regularization, such as dropout (randomly set some outputs to 0, which is actually a random nonlinear activation in disguise)\nEMBEDDING_DROPOUT = 0.00\nMLP_DROPOUT_RATIO = 0.15\nCLASSIFIER_DROPOUT_RATIO = 0.00\n\n# Initiailizers\nINIT_HE_UNIFORM = tf.keras.initializers.he_uniform\nINIT_GLOROT_UNIFORM = tf.keras.initializers.glorot_uniform\nINIT_ZEROS = tf.keras.initializers.constant(0.0)\n# Activations\nGELU = tf.keras.activations.gelu\n\nprint(f'UNITS: {UNITS}')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:45.642707Z","iopub.execute_input":"2023-04-20T20:50:45.64336Z","iopub.status.idle":"2023-04-20T20:50:45.651732Z","shell.execute_reply.started":"2023-04-20T20:50:45.643319Z","shell.execute_reply":"2023-04-20T20:50:45.650702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transformer\n\nNeed to implement transformer from scratch as TFLite does not support the native TF implementation of MultiHeadAttention. Start implementing the Transformer model (a neural network architecture used in natural language processing tasks), which includes the MultiHeadAttention layer. Since the MultiHeadAttention layer is a key component of the Transformer model, it may not be possible to use a pre-existing Transformer implementation in TensorFlow in TFLite if its implementation is not supported by TFLite. Therefore, you need to write the implementation code of Transformer yourself instead of relying on the implementation of the MultiHeadAttention layer in TensorFlow.","metadata":{}},{"cell_type":"markdown","source":"<img src=\"https://pic4.zhimg.com/v2-f6380627207ff4d1e72addfafeaff0bb_r.jpg\">","metadata":{}},{"cell_type":"markdown","source":"Encoder: The input is the Embedding of the word, plus the position encoding, and then enters a unified structure, which can be looped many times (N times), that is to say, there are many layers (N layers). Each layer can be divided into an Attention layer and a fully connected layer, and some additional processing is added, such as Skip Connection, to make a skip connection, and then a Normalization layer is added. In fact, its own model is still very simple.\nDecoder: The first input is the prefix information, and the subsequent one is the Embedding produced last time, adding the position code, and then entering a module that can be repeated many times. This module can be divided into three parts. The first part is also the Attention layer, the second part is cross Attention, not Self-Attention, and the third part is the fully connected layer. Also used skip connection and Normalization.\nOutput: The final output must pass through the Linear layer (full connection layer), and then predict through softmax.","metadata":{}},{"cell_type":"markdown","source":"The Encoder part is a stack of N identical structures, and each structure can be subdivided into the following structures:\n\n1. Perform embedding (word embedding) on ​​the input one-hot encoded samples\n2. Add location code\n3. Introduce self-attention of multi-head mechanism\n4. Add the input and output of self-attention (residual network structure)\n5. Layer Normalization (layer normalization), normalize the data at all times\n6. Feedforword neural network structure\n7. Add the input and output of Feedforword (residual network structure)\n8. Layer Normalization, normalize the data at all times\n9. Repeat the structure of N layers 3-8","metadata":{}},{"cell_type":"markdown","source":"The Decoder part is also a stack of N identical structures, and each structure can be subdivided into the following structures:\n\n1. Perform embedding (word embedding) on ​​the input one-hot encoded samples\n2. Add location code\n3. Introduce self-attention of multi-head mechanism\n4. Add the input and output of self-attention (residual network structure)\n5. Layer Normalization,\n6. Standardize the data at all times and use the value obtained in the previous step as the value, and perform Self-Attenton with q and k obtained from the encoder\n7. Add the input and output of self-attention (residual network structure)\n8. Layer Normalization (layer normalization), normalize the data at all times\n9. Feedforword neural network (Feedforword) structure\n10. Add the input and output of Feedforword (residual network structure)\n11. Layer Normalization, standardize the data at all times\n12. Repeat the structure of N layer 3-11","metadata":{}},{"cell_type":"code","source":"# based on: https://stackoverflow.com/\n#questions/67342988/verifying-the-implementation-of-multihead-attention-in-transformer\n# replaced softmax with softmax layer to support masked softmax\n\n#scaled dot-product attention is an Attention mechanism in the Transformer model, which is a method of calculating Attention weights.\n#In this method, the dot product of Query and Key is divided by a scaling factor, then normalized by the softmax function, and finally multiplied by Value to obtain the Attention output\ndef scaled_dot_product(q,k,v, softmax, attention_mask):\n    #calculates Q . K(transpose)\n    qkt = tf.matmul(q,k,transpose_b=True)\n    #caculates scaling factor\n    dk = tf.math.sqrt(tf.cast(q.shape[-1],dtype=tf.float32))\n    scaled_qkt = qkt/dk\n    softmax = softmax(scaled_qkt, mask=attention_mask)\n    \n    z = tf.matmul(softmax,v)\n    #shape: (m,Tx,depth), same shape as q,k,v\n    return z\n\nclass MultiHeadAttention(tf.keras.layers.Layer):\n    def __init__(self,d_model,num_of_heads):\n        super(MultiHeadAttention,self).__init__()\n        self.d_model = d_model\n        self.num_of_heads = num_of_heads\n        self.depth = d_model//num_of_heads\n        self.wq = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wk = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wv = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wo = tf.keras.layers.Dense(d_model)\n        self.softmax = tf.keras.layers.Softmax()\n        \n    def call(self,x, attention_mask):\n        \n        multi_attn = []\n        for i in range(self.num_of_heads):\n            Q = self.wq[i](x)\n            K = self.wk[i](x)\n            V = self.wv[i](x)\n            multi_attn.append(scaled_dot_product(Q,K,V, self.softmax, attention_mask))\n            \n        multi_head = tf.concat(multi_attn,axis=-1)\n        multi_head_attention = self.wo(multi_head)\n        return multi_head_attention","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:45.653367Z","iopub.execute_input":"2023-04-20T20:50:45.653727Z","iopub.status.idle":"2023-04-20T20:50:45.666901Z","shell.execute_reply.started":"2023-04-20T20:50:45.653691Z","shell.execute_reply":"2023-04-20T20:50:45.665918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code is an implementation of MultiHeadAttention. It linearly transforms the input tensor x through multiple Dense layers, and then passes the transformed tensors into the scaled_dot_product function as Q, K, and V respectively, and calculates the output of the multi-head attention mechanism. Finally, the output of the multi-head attention mechanism is spliced ​​together, and then linearly transformed through a Dense layer to obtain the final output multi_head_attention. The scaled_dot_product function is a function to calculate Q.K^T, where Q, K, and V are query, key, and value matrices, respectively, and attention_mask is a tensor used for the mask. The softmax function is a function used to calculate the softmax value.","metadata":{}},{"cell_type":"markdown","source":"# transformer core architecture","metadata":{}},{"cell_type":"markdown","source":"<img src=\"https://4143056590-files.gitbook.io/~/files/v0/b/gitbook-legacy-files/o/assets%2F-LpO5sn2FY1C9esHFJmo%2F-M1uVIrSPBnanwyeV0ps%2F-M1uVKtDCvJ7TGjfZPuP%2Fencoder-decoder-2.jpg?generation=1583677008527428&alt=media\">","metadata":{}},{"cell_type":"code","source":"# Full Transformer\nclass Transformer(tf.keras.Model):\n    def __init__(self, num_blocks):\n        super(Transformer, self).__init__(name='transformer')\n        self.num_blocks = num_blocks\n    \n    def build(self, input_shape):\n        self.ln_1s = []\n        self.mhas = []\n        self.ln_2s = []\n        self.mlps = []\n        # Make Transformer Blocks\n        for i in range(self.num_blocks):\n            # Multi Head Attention\n            self.mhas.append(MultiHeadAttention(UNITS, 8))\n            # Multi Layer Perception\n            self.mlps.append(tf.keras.Sequential([\n                tf.keras.layers.Dense(UNITS * MLP_RATIO, activation=GELU, kernel_initializer=INIT_GLOROT_UNIFORM),\n                tf.keras.layers.Dropout(MLP_DROPOUT_RATIO),\n                tf.keras.layers.Dense(UNITS, kernel_initializer=INIT_HE_UNIFORM),\n            ]))\n        \n    def call(self, x, attention_mask):\n        # Iterate input over transformer blocks\n        for mha, mlp in zip(self.mhas, self.mlps):\n            x = x + mha(x, attention_mask)\n            x = x + mlp(x)\n    \n        return x","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:45.670055Z","iopub.execute_input":"2023-04-20T20:50:45.670359Z","iopub.status.idle":"2023-04-20T20:50:45.681546Z","shell.execute_reply.started":"2023-04-20T20:50:45.670317Z","shell.execute_reply":"2023-04-20T20:50:45.680613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is a Python Transformer model, which is a class inherited from tf.keras.Model. It has a constructor where num_blocks is an integer representing the number of Transformer blocks. In the build function, it creates multiple Multi Head Attention and Multi Layer Perception objects and stores them in class variables. In the call function, it iterates over the input data and passes it to each Transformer block. Each block contains a Multi Head Attention and a Multi Layer Perception layer. The purpose of this model is to implement natural language processing tasks, such as machine translation, text summarization, etc.","metadata":{}},{"cell_type":"markdown","source":"# Landmark Embedding\nKey point embedding\", where \"Landmark\" represents the key points of the face, and \"Embedding\" represents the process of mapping these key point information to a low-dimensional vector space. Therefore, the Chinese meaning of \"Landmark Embedding\" can be understood as \"the face Embedding keypoint information into a low-dimensional vector space\".","metadata":{}},{"cell_type":"markdown","source":"Landmark Embedding is a method of converting facial key point information into a low-dimensional vector representation. In tasks such as face recognition and facial expression recognition, Landmark Embedding is usually used to extract facial feature representations.\n\nSpecifically, Landmark Embedding maps the key point coordinates in the face image to a vector representation in a low-dimensional space. This vector representation can contain information about face shape, pose, and expression, and can be used to compare similarities or differences between different faces. Compared with directly using pixel information or high-dimensional feature vector representation, Landmark Embedding can improve the accuracy and robustness of face recognition and expression recognition.","metadata":{}},{"cell_type":"code","source":"class LandmarkEmbedding(tf.keras.Model):\n    def __init__(self, units, name):\n        super(LandmarkEmbedding, self).__init__(name=f'{name}_embedding')\n        self.units = units\n        \n    def build(self, input_shape):\n        # Embedding for missing landmark in frame, initizlied with zeros\n        self.empty_embedding = self.add_weight(\n            name=f'{self.name}_empty_embedding',\n            shape=[self.units],\n            initializer=INIT_ZEROS,\n        )\n        # Embedding\n        self.dense = tf.keras.Sequential([\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM),\n            tf.keras.layers.Activation(GELU),\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name=f'{self.name}_dense')\n\n    def call(self, x):\n        return tf.where(\n                # Checks whether landmark is missing in frame\n                tf.reduce_sum(x, axis=2, keepdims=True) == 0,\n                # If so, the empty embedding is used\n                self.empty_embedding,\n                # Otherwise the landmark data is embedded\n                self.dense(x),\n            )","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:45.682821Z","iopub.execute_input":"2023-04-20T20:50:45.683391Z","iopub.status.idle":"2023-04-20T20:50:45.697014Z","shell.execute_reply.started":"2023-04-20T20:50:45.683351Z","shell.execute_reply":"2023-04-20T20:50:45.69593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Embedding","metadata":{}},{"cell_type":"code","source":"class Embedding(tf.keras.Model):\n    def __init__(self):\n        super(Embedding, self).__init__()\n        \n    def get_diffs(self, l):\n        S = l.shape[2]\n        other = tf.expand_dims(l, 3)\n        other = tf.repeat(other, S, axis=3)\n        other = tf.transpose(other, [0,1,3,2])\n        diffs = tf.expand_dims(l, 3) - other\n        diffs = tf.reshape(diffs, [-1, INPUT_SIZE, S*S])\n        return diffs\n\n    def build(self, input_shape):\n        # Positional Embedding, initialized with zeros\n        self.positional_embedding = tf.keras.layers.Embedding(INPUT_SIZE+1, UNITS, embeddings_initializer=INIT_ZEROS)\n        # Embedding layer for Landmarks\n        self.lips_embedding = LandmarkEmbedding(LIPS_UNITS, 'lips')\n        self.left_hand_embedding = LandmarkEmbedding(HANDS_UNITS, 'left_hand')\n        self.pose_embedding = LandmarkEmbedding(POSE_UNITS, 'pose')\n        # Landmark Weights\n        self.landmark_weights = tf.Variable(tf.zeros([3], dtype=tf.float32), name='landmark_weights')\n        # Fully Connected Layers for combined landmarks\n        self.fc = tf.keras.Sequential([\n            tf.keras.layers.Dense(UNITS, name='fully_connected_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM),\n            tf.keras.layers.Activation(GELU),\n            tf.keras.layers.Dense(UNITS, name='fully_connected_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name='fc')\n\n\n    def call(self, lips0, left_hand0, pose0, non_empty_frame_idxs, training=False):\n        # Lips\n        lips_embedding = self.lips_embedding(lips0)\n        # Left Hand\n        left_hand_embedding = self.left_hand_embedding(left_hand0)\n        # Pose\n        pose_embedding = self.pose_embedding(pose0)\n        # Merge Embeddings of all landmarks with mean pooling\n        x = tf.stack((\n            lips_embedding, left_hand_embedding, pose_embedding,\n        ), axis=3)\n        x = x * tf.nn.softmax(self.landmark_weights)\n        x = tf.reduce_sum(x, axis=3)\n        # Fully Connected Layers\n        x = self.fc(x)\n        # Add Positional Embedding\n        max_frame_idxs = tf.clip_by_value(\n                tf.reduce_max(non_empty_frame_idxs, axis=1, keepdims=True),\n                1,\n                np.PINF,\n            )\n        normalised_non_empty_frame_idxs = tf.where(\n            tf.math.equal(non_empty_frame_idxs, -1.0),\n            INPUT_SIZE,\n            tf.cast(\n                non_empty_frame_idxs / max_frame_idxs * INPUT_SIZE,\n                tf.int32,\n            ),\n        )\n        x = x + self.positional_embedding(normalised_non_empty_frame_idxs)\n        \n        return x","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:45.699462Z","iopub.execute_input":"2023-04-20T20:50:45.700502Z","iopub.status.idle":"2023-04-20T20:50:45.715289Z","shell.execute_reply.started":"2023-04-20T20:50:45.700476Z","shell.execute_reply":"2023-04-20T20:50:45.714357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Augmentation data enhancement","metadata":{}},{"cell_type":"markdown","source":"The function of the add_noise method is to replace all 0 in the input tensor t with 0, and add a random noise that obeys a normal distribution to all non-zero elements. The standard deviation of this noise is noise_std, which is defined in the constructor of the class. This method uses TensorFlow's tf.where function, which takes three arguments: a boolean tensor, a tensor x, and a tensor y. If the element in the Boolean tensor is True, return the element at the corresponding position in x; otherwise return the element at the corresponding position in y. In this method, if the element in t is 0, it returns 0; otherwise, it returns t plus a normal distribution of random noise.\nIf the train flag is True, noise will be added on each tensor.","metadata":{}},{"cell_type":"code","source":"# Not used, adds random X/y translation to input on samples level\nclass Augmentation(tf.keras.layers.Layer):\n    def __init__(self, noise_std):\n        super(Augmentation, self).__init__()\n        self.noise_std = noise_std\n    \n    def add_noise(self, t):\n        B = tf.shape(t)[0]\n        return tf.where(\n            t == 0.0,\n            0.0,\n            t + tf.random.normal([B,1,1,tf.shape(t)[3]], 0, self.noise_std),\n        )\n    \n    def call(self, lips0, left_hand0, pose0, training=False):\n        if training:\n            # Lips\n            lips0 = self.add_noise(lips0)\n            # Left Hand\n            left_hand0 = self.add_noise(left_hand0)\n            # Pose\n            pose0 = self.add_noise(pose0)\n        \n        return lips0, left_hand0, pose0","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:45.718352Z","iopub.execute_input":"2023-04-20T20:50:45.718747Z","iopub.status.idle":"2023-04-20T20:50:45.729076Z","shell.execute_reply.started":"2023-04-20T20:50:45.71872Z","shell.execute_reply":"2023-04-20T20:50:45.728073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sparse Categorical Crossentropy With Label Smoothing","metadata":{}},{"cell_type":"code","source":"# source:: https://stackoverflow.com/questions/60689185/label-smoothing-for-sparse-categorical-crossentropy\ndef scce_with_ls(y_true, y_pred):\n    # One Hot Encode Sparsely Encoded Target Sign\n    y_true = tf.cast(y_true, tf.int32)\n    y_true = tf.one_hot(y_true, NUM_CLASSES, axis=1)\n    y_true = tf.squeeze(y_true, axis=2)\n    # Categorical Crossentropy with native label smoothing support\n    return tf.keras.losses.categorical_crossentropy(y_true, y_pred, label_smoothing=0.25)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:45.731856Z","iopub.execute_input":"2023-04-20T20:50:45.73248Z","iopub.status.idle":"2023-04-20T20:50:45.741391Z","shell.execute_reply.started":"2023-04-20T20:50:45.732443Z","shell.execute_reply":"2023-04-20T20:50:45.740441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"markdown","source":"This code is an implementation of a TensorFlow model.\nIt has two inputs, \"frames\" and \"non_empty_frame_idxs\".\nIn this model, frames is a video data containing multiple frames, and non_empty_frame_idxs indicates which frames in frames have content.\nIn this code, locations with valid frames are selected by masking so that only those frames are trained.\nThis model uses the Transformer architecture, which processes input by combining multiple layers with attention mechanisms.\nIn this model, each frame is embedded into three different representations, namely LIPS, LEFT HAND and POSE, which are used to construct the input of the Transformer.\nIn this model, some additional tricks are also implemented, such as random frame masking, category loss (losing some features when classifying), and label smoothing, etc.\nFinally, the model also includes an optimizer and some evaluation metrics. The optimizer uses AdamW, and the evaluation metrics include sparse classification accuracy, top-k accuracy for sparse classification, etc.","metadata":{}},{"cell_type":"code","source":"def get_model():\n    # Inputs\n    frames = tf.keras.layers.Input([INPUT_SIZE, N_COLS, N_DIMS], dtype=tf.float32, name='frames')\n    non_empty_frame_idxs = tf.keras.layers.Input([INPUT_SIZE], dtype=tf.float32, name='non_empty_frame_idxs')\n    # Padding Mask\n    mask0 = tf.cast(tf.math.not_equal(non_empty_frame_idxs, -1), tf.float32)\n    mask0 = tf.expand_dims(mask0, axis=2)\n    # Random Frame Masking\n    mask = tf.where(\n        (tf.random.uniform(tf.shape(mask0)) > 0.25) & tf.math.not_equal(mask0, 0.0),\n        1.0,\n        0.0,\n    )\n    # Correct Samples Which are all masked now...\n    mask = tf.where(\n        tf.math.equal(tf.reduce_sum(mask, axis=[1,2], keepdims=True), 0.0),\n        mask0,\n        mask,\n    )\n    \n    \n    \"\"\"\n        left_hand: 468:489\n        pose: 489:522\n        right_hand: 522:543\n    \"\"\"\n    x = frames\n    x = tf.slice(x, [0,0,0,0], [-1,INPUT_SIZE, N_COLS, 2])\n    # LIPS\n    lips = tf.slice(x, [0,0,LIPS_START,0], [-1,INPUT_SIZE, 40, 2])\n    lips = tf.where(\n            tf.math.equal(lips, 0.0),\n            0.0,\n            (lips - LIPS_MEAN) / LIPS_STD,\n        )\n    # LEFT HAND\n    left_hand = tf.slice(x, [0,0,40,0], [-1,INPUT_SIZE, 21, 2])\n    left_hand = tf.where(\n            tf.math.equal(left_hand, 0.0),\n            0.0,\n            (left_hand - LEFT_HANDS_MEAN) / LEFT_HANDS_STD,\n        )\n    # POSE\n    pose = tf.slice(x, [0,0,61,0], [-1,INPUT_SIZE, 5, 2])\n    pose = tf.where(\n            tf.math.equal(pose, 0.0),\n            0.0,\n            (pose - POSE_MEAN) / POSE_STD,\n        )\n    \n    # Flatten\n    lips = tf.reshape(lips, [-1, INPUT_SIZE, 40*2])\n    left_hand = tf.reshape(left_hand, [-1, INPUT_SIZE, 21*2])\n    pose = tf.reshape(pose, [-1, INPUT_SIZE, 5*2])\n        \n    # Embedding\n    x = Embedding()(lips, left_hand, pose, non_empty_frame_idxs)\n    \n    # Encoder Transformer Blocks\n    x = Transformer(NUM_BLOCKS)(x, mask)\n    \n    # Pooling\n    x = tf.reduce_sum(x * mask, axis=1) / tf.reduce_sum(mask, axis=1)\n    # Classifier Dropout\n    x = tf.keras.layers.Dropout(CLASSIFIER_DROPOUT_RATIO)(x)\n    # Classification Layer\n    x = tf.keras.layers.Dense(NUM_CLASSES, activation=tf.keras.activations.softmax, kernel_initializer=INIT_GLOROT_UNIFORM)(x)\n    \n    outputs = x\n    \n    # Create Tensorflow Model\n    model = tf.keras.models.Model(inputs=[frames, non_empty_frame_idxs], outputs=outputs)\n    \n    # Sparse Categorical Cross Entropy With Label Smoothing\n    loss = scce_with_ls\n    #SGDW is an optimizer based on SGD but with the concept of momentum added.\n    #The role of momentum is to not only subtract the gradient of the current iteration, but also subtract the weighted sum of the gradient of the previous t-1 iteration when updating the parameters.\n    #The advantage of doing this is that it can make the parameter update smoother and avoid the shock during the parameter update process.\n    #optimizer = tfa.optimizers.SGDW(\n    #learning_rate=lr, weight_decay=wd, momentum=0.9)\n    #optimizer = tf.keras.optimizers.SGD(lr=0.001, momentum=0.0, nesterov=False)\n    #optimizer = tfa.optimizers.SGDW(learning_rate=0.001, momentum=0.7, weight_decay=0.005)\n    #Adam Optimizer with weight decay\n    optimizer = tfa.optimizers.AdamW(learning_rate=1e-3, weight_decay=1e-5, clipnorm=1.0)\n    #The learning rate is 1e-3, the weight decay is 1e-5, and the gradient clipping threshold is 1.0    # TopK Metrics\n    metrics = [\n        tf.keras.metrics.SparseCategoricalAccuracy(name='acc'),\n        tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5, name='top_5_acc'),\n        tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10, name='top_10_acc'),\n    ]\n    \n    model.compile(loss=loss, optimizer=optimizer, metrics=metrics)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:45.744436Z","iopub.execute_input":"2023-04-20T20:50:45.744729Z","iopub.status.idle":"2023-04-20T20:50:45.762496Z","shell.execute_reply.started":"2023-04-20T20:50:45.744704Z","shell.execute_reply":"2023-04-20T20:50:45.761453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nmodel = get_model()","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:45.764031Z","iopub.execute_input":"2023-04-20T20:50:45.76436Z","iopub.status.idle":"2023-04-20T20:50:49.241064Z","shell.execute_reply.started":"2023-04-20T20:50:45.764325Z","shell.execute_reply":"2023-04-20T20:50:49.240014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot model summary\nmodel.summary(expand_nested=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:49.242745Z","iopub.execute_input":"2023-04-20T20:50:49.243128Z","iopub.status.idle":"2023-04-20T20:50:49.416677Z","shell.execute_reply.started":"2023-04-20T20:50:49.243085Z","shell.execute_reply":"2023-04-20T20:50:49.415927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(model, show_shapes=True, show_dtype=True, show_layer_names=True, expand_nested=True, show_layer_activations=True)","metadata":{"_kg_hide-input":false,"scrolled":true,"execution":{"iopub.status.busy":"2023-04-20T20:50:49.417716Z","iopub.execute_input":"2023-04-20T20:50:49.418418Z","iopub.status.idle":"2023-04-20T20:50:50.385771Z","shell.execute_reply.started":"2023-04-20T20:50:49.418379Z","shell.execute_reply":"2023-04-20T20:50:50.384688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# No NaN Predictions","metadata":{}},{"cell_type":"code","source":"if not PREPROCESS_DATA and TRAIN_MODEL:\n    y_pred = model.predict_on_batch(X_batch).flatten()\n\n    print(f'# NaN Values In Prediction: {np.isnan(y_pred).sum()}')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:50.38791Z","iopub.execute_input":"2023-04-20T20:50:50.388552Z","iopub.status.idle":"2023-04-20T20:50:54.470866Z","shell.execute_reply.started":"2023-04-20T20:50:50.388508Z","shell.execute_reply":"2023-04-20T20:50:54.469615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weight Initialization","metadata":{}},{"cell_type":"code","source":"if not PREPROCESS_DATA and TRAIN_MODEL:\n    plt.figure(figsize=(12,5))\n    plt.title(f'Softmax Output Initialized Model | µ={y_pred.mean():.3f}, σ={y_pred.std():.3f}', pad=25)\n    pd.Series(y_pred).plot(kind='hist', bins=128, label='Class Probability')\n    plt.xlim(0, max(y_pred) * 1.1)\n    plt.vlines([1 / NUM_CLASSES], 0, plt.ylim()[1], color='red', label=f'Random Guessing Baseline 1/NUM_CLASSES={1 / NUM_CLASSES:.3f}')\n    plt.grid()\n    plt.legend()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:54.472457Z","iopub.execute_input":"2023-04-20T20:50:54.473531Z","iopub.status.idle":"2023-04-20T20:50:55.012915Z","shell.execute_reply.started":"2023-04-20T20:50:54.473485Z","shell.execute_reply":"2023-04-20T20:50:55.011825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Learning Rate Scheduler","metadata":{}},{"cell_type":"code","source":"def lrfn(current_step, num_warmup_steps, lr_max, num_cycles=0.50, num_training_steps=N_EPOCHS):\n    \n    if current_step < num_warmup_steps:\n        if WARMUP_METHOD == 'log':\n            return lr_max * 0.10 ** (num_warmup_steps - current_step)\n        else:\n            return lr_max * 2 ** -(num_warmup_steps - current_step)\n    else:\n        progress = float(current_step - num_warmup_steps) / float(max(1, num_training_steps - num_warmup_steps))\n\n        return max(0.0, 0.5 * (1.0 + math.cos(math.pi * float(num_cycles) * 2.0 * progress))) * lr_max","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:55.014459Z","iopub.execute_input":"2023-04-20T20:50:55.017576Z","iopub.status.idle":"2023-04-20T20:50:55.024629Z","shell.execute_reply.started":"2023-04-20T20:50:55.017542Z","shell.execute_reply":"2023-04-20T20:50:55.023333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_lr_schedule(lr_schedule, epochs):\n    fig = plt.figure(figsize=(20, 10))\n    plt.plot([None] + lr_schedule + [None])\n    # X Labels\n    x = np.arange(1, epochs + 1)\n    x_axis_labels = [i if epochs <= 40 or i % 5 == 0 or i == 1 else None for i in range(1, epochs + 1)]\n    plt.xlim([1, epochs])\n    plt.xticks(x, x_axis_labels) # set tick step to 1 and let x axis start at 1\n    \n    # Increase y-limit for better readability\n    plt.ylim([0, max(lr_schedule) * 1.1])\n    \n    # Title\n    schedule_info = f'start: {lr_schedule[0]:.1E}, max: {max(lr_schedule):.1E}, final: {lr_schedule[-1]:.1E}'\n    plt.title(f'Step Learning Rate Schedule, {schedule_info}', size=18, pad=12)\n    \n    # Plot Learning Rates\n    for x, val in enumerate(lr_schedule):\n        if epochs <= 40 or x % 5 == 0 or x is epochs - 1:\n            if x < len(lr_schedule) - 1:\n                if lr_schedule[x - 1] < val:\n                    ha = 'right'\n                else:\n                    ha = 'left'\n            elif x == 0:\n                ha = 'right'\n            else:\n                ha = 'left'\n            plt.plot(x + 1, val, 'o', color='black');\n            offset_y = (max(lr_schedule) - min(lr_schedule)) * 0.02\n            plt.annotate(f'{val:.1E}', xy=(x + 1, val + offset_y), size=12, ha=ha)\n    \n    plt.xlabel('Epoch', size=16, labelpad=5)\n    plt.ylabel('Learning Rate', size=16, labelpad=5)\n    plt.grid()\n    plt.show()\n\n# Learning rate for encoder\nLR_SCHEDULE = [lrfn(step, num_warmup_steps=N_WARMUP_EPOCHS, lr_max=LR_MAX, num_cycles=0.50) for step in range(N_EPOCHS)]\n# Plot Learning Rate Schedule\nplot_lr_schedule(LR_SCHEDULE, epochs=N_EPOCHS)\n# Learning Rate Callback\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lambda step: LR_SCHEDULE[step], verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:55.026586Z","iopub.execute_input":"2023-04-20T20:50:55.027043Z","iopub.status.idle":"2023-04-20T20:50:55.77083Z","shell.execute_reply.started":"2023-04-20T20:50:55.027006Z","shell.execute_reply":"2023-04-20T20:50:55.769851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weight Decay Callback","metadata":{}},{"cell_type":"code","source":"# Custom callback to update weight decay with learning rate\nclass WeightDecayCallback(tf.keras.callbacks.Callback):\n    def __init__(self, wd_ratio=WD_RATIO):\n        self.step_counter = 0\n        self.wd_ratio = wd_ratio\n    \n    def on_epoch_begin(self, epoch, logs=None):\n        model.optimizer.weight_decay = model.optimizer.learning_rate * self.wd_ratio\n        print(f'learning rate: {model.optimizer.learning_rate.numpy():.2e}, weight decay: {model.optimizer.weight_decay.numpy():.2e}')","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:55.772405Z","iopub.execute_input":"2023-04-20T20:50:55.773565Z","iopub.status.idle":"2023-04-20T20:50:55.78041Z","shell.execute_reply.started":"2023-04-20T20:50:55.773521Z","shell.execute_reply":"2023-04-20T20:50:55.779413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Performance Benchmark","metadata":{}},{"cell_type":"code","source":"%%timeit -n 100\nif TRAIN_MODEL:\n    # Verify model prediction is <<<100ms\n    model.predict_on_batch({ 'frames': X_train[:1], 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_TRAIN[:1] })\n    pass","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:50:55.781818Z","iopub.execute_input":"2023-04-20T20:50:55.782256Z","iopub.status.idle":"2023-04-20T20:51:13.8806Z","shell.execute_reply.started":"2023-04-20T20:50:55.782218Z","shell.execute_reply":"2023-04-20T20:51:13.879421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"if USE_VAL:\n    # Verify Validation Dataset Covers All Signs\n    print(f'# Unique Signs in Validation Set: {pd.Series(y_val).nunique()}')\n    # Value Counts\n    display(pd.Series(y_val).value_counts().to_frame('Count').iloc[[1,2,3,-3,-2,-1]])","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:51:13.88257Z","iopub.execute_input":"2023-04-20T20:51:13.882951Z","iopub.status.idle":"2023-04-20T20:51:13.888686Z","shell.execute_reply.started":"2023-04-20T20:51:13.882911Z","shell.execute_reply":"2023-04-20T20:51:13.887411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluate Initialzied Model","metadata":{}},{"cell_type":"code","source":"# Sanity Check\nif TRAIN_MODEL and USE_VAL:\n    _ = model.evaluate(*validation_data, verbose=2)","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:51:13.890451Z","iopub.execute_input":"2023-04-20T20:51:13.891146Z","iopub.status.idle":"2023-04-20T20:51:13.899547Z","shell.execute_reply.started":"2023-04-20T20:51:13.891106Z","shell.execute_reply":"2023-04-20T20:51:13.898846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"if TRAIN_MODEL:\n    # Clear all models in GPU\n    tf.keras.backend.clear_session()\n\n    # Get new fresh model\n    model = get_model()\n    \n    # Sanity Check\n    model.summary()\n\n    # Actual Training\n    history = model.fit(\n            x=get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN),\n            steps_per_epoch=len(X_train) // (NUM_CLASSES * BATCH_ALL_SIGNS_N),\n            epochs=N_EPOCHS,\n            # Only used for validation data since training data is a generator\n            #\"Only for validation data, since training data is a generator\".\n            #If the generator is used to train the model, the validation data must be used to evaluate the performance of the model.\n            # This is because the generator produces new data in each epoch instead of loading all the data into memory.\n            # Therefore, the training data cannot be used during training to evaluate the performance of the model. Instead, the validation data must be used to evaluate the performance of the model.\n            batch_size=BATCH_SIZE,\n            validation_data=validation_data,\n            callbacks=[\n                lr_callback,\n                WeightDecayCallback(),\n            ],\n            verbose = 2,\n        )","metadata":{"execution":{"iopub.status.busy":"2023-04-20T20:51:13.901022Z","iopub.execute_input":"2023-04-20T20:51:13.9017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"model.fit() is a method in TensorFlow that is used to train a model on a given dataset. It accepts multiple parameters such as training data, validation data, number of epochs, batch size, etc. It trains the model by using an optimizer to minimize a loss function. A loss function is a way to measure how well a model performs at predicting an output. The optimizer adjusts the weights of the model to minimize this loss function. During training, model.fit() prints out metrics like loss and accuracy","metadata":{}},{"cell_type":"code","source":"# Save Model Weights\nmodel.save_weights('model.h5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if USE_VAL:\n    # Validation Predictions\n    y_val_pred = model.predict({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, verbose=2).argmax(axis=1)\n    # Label\n    labels = [ORD2SIGN.get(i).replace(' ', '_') for i in range(NUM_CLASSES)]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Landmark Attention Weights","metadata":{}},{"cell_type":"code","source":"# Landmark Weights\nfor w in model.get_layer('embedding').weights:\n    if 'landmark_weights' in w.name:\n        weights = scipy.special.softmax(w)\n\nlandmarks = ['lips_embedding', 'left_hand_embedding', 'pose_embedding']\n\nfor w, lm in zip(weights, landmarks):\n    print(f'{lm} weight: {(w*100):.1f}%')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Classification Report","metadata":{}},{"cell_type":"code","source":"def print_classification_report():\n    # Classification report for all signs\n    classification_report = sklearn.metrics.classification_report(\n            y_val,\n            y_val_pred,\n            target_names=labels,\n            output_dict=True,\n        )\n    # Round Data for better readability\n    classification_report = pd.DataFrame(classification_report).T\n    classification_report = classification_report.round(2)\n    classification_report = classification_report.astype({\n            'support': np.uint16,\n        })\n    # Add signs\n    classification_report['sign'] = [e if e in SIGN2ORD else -1 for e in classification_report.index]\n    classification_report['sign_ord'] = classification_report['sign'].apply(SIGN2ORD.get).fillna(-1).astype(np.int16)\n    # Sort on F1-score\n    classification_report = pd.concat((\n        classification_report.head(NUM_CLASSES).sort_values('f1-score', ascending=False),\n        classification_report.tail(3),\n    ))\n\n    pd.options.display.max_rows = 999\n    display(classification_report)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if USE_VAL:\n    print_classification_report()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training History","metadata":{}},{"cell_type":"markdown","source":"draw the training curve","metadata":{}},{"cell_type":"code","source":"def plot_history_metric(metric, f_best=np.argmax, ylim=None, yscale=None, yticks=None):\n    plt.figure(figsize=(20, 10))\n    \n    values = history.history[metric]\n    N_EPOCHS = len(values)\n    val = 'val' in ''.join(history.history.keys())\n    # Epoch Ticks\n    if N_EPOCHS <= 20:\n        x = np.arange(1, N_EPOCHS + 1)\n    else:\n        x = [1, 5] + [10 + 5 * idx for idx in range((N_EPOCHS - 10) // 5 + 1)]\n\n    x_ticks = np.arange(1, N_EPOCHS+1)\n\n    # Validation\n    if val:\n        val_values = history.history[f'val_{metric}']\n        val_argmin = f_best(val_values)\n        plt.plot(x_ticks, val_values, label=f'val')\n\n    # summarize history for accuracy\n    plt.plot(x_ticks, values, label=f'train')\n    argmin = f_best(values)\n    plt.scatter(argmin + 1, values[argmin], color='red', s=75, marker='o', label=f'train_best')\n    if val:\n        plt.scatter(val_argmin + 1, val_values[val_argmin], color='purple', s=75, marker='o', label=f'val_best')\n\n    plt.title(f'Model {metric}', fontsize=24, pad=10)\n    plt.ylabel(metric, fontsize=20, labelpad=10)\n\n    if ylim:\n        plt.ylim(ylim)\n\n    if yscale is not None:\n        plt.yscale(yscale)\n        \n    if yticks is not None:\n        plt.yticks(yticks, fontsize=16)\n\n    plt.xlabel('epoch', fontsize=20, labelpad=10)        \n    plt.tick_params(axis='x', labelsize=8)\n    plt.xticks(x, fontsize=16) # set tick step to 1 and let x axis start at 1\n    plt.yticks(fontsize=16)\n    \n    plt.legend(prop={'size': 10})\n    plt.grid()\n    plt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('loss', f_best=np.argmin)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('top_5_acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('top_10_acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission\n\nSubmission code loosley based on [this notebook](https://www.kaggle.com/code/dschettler8845/gislr-learn-eda-baseline#baseline) by [Darien Schettler\n](https://www.kaggle.com/dschettler8845)","metadata":{}},{"cell_type":"code","source":"# TFLite model for submission\nclass TFLiteModel(tf.Module):\n    def __init__(self, model):\n        super(TFLiteModel, self).__init__()\n\n        # Load the feature generation and main models\n        self.preprocess_layer = preprocess_layer\n        self.model = model\n    \n    @tf.function(input_signature=[tf.TensorSpec(shape=[None, N_ROWS, N_DIMS], dtype=tf.float32, name='inputs')])\n    def __call__(self, inputs):\n        # Preprocess Data\n        x, non_empty_frame_idxs = self.preprocess_layer(inputs)\n        # Add Batch Dimension\n        x = tf.expand_dims(x, axis=0)\n        non_empty_frame_idxs = tf.expand_dims(non_empty_frame_idxs, axis=0)\n        # Make Prediction\n        outputs = self.model({ 'frames': x, 'non_empty_frame_idxs': non_empty_frame_idxs })\n        # Squeeze Output 1x250 -> 250\n        outputs = tf.squeeze(outputs, axis=0)\n\n        # Return a dictionary with the output tensor\n        return {'outputs': outputs}\n\n# Define TF Lite Model\ntflite_keras_model = TFLiteModel(model)\n\n# Sanity Check\ndemo_raw_data = load_relevant_data_subset(train['file_path'].values[5])\nprint(f'demo_raw_data shape: {demo_raw_data.shape}, dtype: {demo_raw_data.dtype}')\ndemo_output = tflite_keras_model(demo_raw_data)[\"outputs\"]\nprint(f'demo_output shape: {demo_output.shape}, dtype: {demo_output.dtype}')\ndemo_prediction = demo_output.numpy().argmax()\nprint(f'demo_prediction: {demo_prediction}, correct: {train.iloc[0][\"sign_ord\"]}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Model Converter\nkeras_model_converter = tf.lite.TFLiteConverter.from_keras_model(tflite_keras_model)\n# Convert Model\ntflite_model = keras_model_converter.convert()\n# Write Model\nwith open('/kaggle/working/model.tflite', 'wb') as f:\n    f.write(tflite_model)\n    \n# Zip Model\n!zip submission.zip /kaggle/working/model.tflite","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Verify TFLite model can be loaded and used for prediction\n!pip install tflite-runtime\nimport tflite_runtime.interpreter as tflite\n\ninterpreter = tflite.Interpreter(\"/kaggle/working/model.tflite\")\nfound_signatures = list(interpreter.get_signature_list().keys())\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")\n\noutput = prediction_fn(inputs=demo_raw_data)\nsign = output['outputs'].argmax()\n\nprint(\"PRED : \", ORD2SIGN.get(sign), f'[{sign}]')\nprint(\"TRUE : \", train.sign.values[0], f'[{train.sign_ord.values[0]}]')","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}