{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Improve Best Public Notebook - LB 0.73 => LB 0.76\nThis notebook is a fork of Dengxianxu's public notebook [here][1]. His notebook has LB 0.73. We make 11 changes to boost the CV and LB by `+0.03`\n\n* Train 1 model => Train 4 models\n* Add Time Scale augmentation\n* Ensemble and apply TFLite FP16 quantization\n* Change the following parameters:\n* INPUT_SIZE, 64 => 12\n* BATCH_ALL_SIGNS_N, 4 => 1\n* N_EPOCHS, 250 => 120\n* LANDMARK_UNITS, 384 => 224\n* UNITS, 512 => 376\n* NUM_BLOCKS, 2 => 3\n* MLP_RATIO, 4 => 3\n* MLP_DROPOUT_RATIO, 0.40 => 0.30\n* remove random frame masking\n\n[1]: https://www.kaggle.com/code/dengxianxu/transformer-model-training?scriptVersionId=126792661","metadata":{"papermill":{"duration":0.00973,"end_time":"2023-05-02T02:05:50.617725","exception":false,"start_time":"2023-05-02T02:05:50.607995","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Load Libraries","metadata":{"papermill":{"duration":0.008174,"end_time":"2023-05-02T02:05:50.634565","exception":false,"start_time":"2023-05-02T02:05:50.626391","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sn\n\nfrom tensorflow import keras\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split, GroupShuffleSplit \n\nimport glob\nimport sys\nimport os\nimport math\nimport gc\nimport sys\nimport sklearn\nimport scipy\n\nprint(f'Tensorflow V{tf.__version__}')\nprint(f'Keras V{tf.keras.__version__}')\nprint(f'Python V{sys.version}')","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:05:50.653763Z","iopub.status.busy":"2023-05-02T02:05:50.652725Z","iopub.status.idle":"2023-05-02T02:05:58.526551Z","shell.execute_reply":"2023-05-02T02:05:58.523865Z"},"papermill":{"duration":7.886362,"end_time":"2023-05-02T02:05:58.529365","exception":false,"start_time":"2023-05-02T02:05:50.643003","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Parameters","metadata":{"papermill":{"duration":0.008847,"end_time":"2023-05-02T02:05:58.547388","exception":false,"start_time":"2023-05-02T02:05:58.538541","status":"completed"},"tags":[]}},{"cell_type":"code","source":"PREPROCESS_DATA = False\nTRAIN_MODEL = True\nUSE_VAL = False\n\nN_ROWS = 543\nN_DIMS = 3\nDIM_NAMES = ['x', 'y', 'z']\nSEED = 42\nNUM_CLASSES = 250\nIS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\nVERBOSE = 1 if IS_INTERACTIVE else 2\n\nINPUT_SIZE = 12 # WAS 64\nBATCH_ALL_SIGNS_N = 1 # WAS 4\nBATCH_SIZE = 256 # WAS 1024\nN_EPOCHS = 120 # WAS 250\nLR_MAX = 0.001\nN_WARMUP_EPOCHS = 0\nWD_RATIO = 0.05\nMASK_VAL = 4237\n\nN_MODELS = 4 # WAS 1\nDEBUG = False","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:05:58.567467Z","iopub.status.busy":"2023-05-02T02:05:58.566132Z","iopub.status.idle":"2023-05-02T02:05:58.573569Z","shell.execute_reply":"2023-05-02T02:05:58.572634Z"},"papermill":{"duration":0.018948,"end_time":"2023-05-02T02:05:58.575554","exception":false,"start_time":"2023-05-02T02:05:58.556606","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prints Shape and Dtype For List Of Variables\ndef print_shape_dtype(l, names):\n    for e, n in zip(l, names):\n        print(f'{n} shape: {e.shape}, dtype: {e.dtype}')","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:05:58.593972Z","iopub.status.busy":"2023-05-02T02:05:58.593175Z","iopub.status.idle":"2023-05-02T02:05:58.598320Z","shell.execute_reply":"2023-05-02T02:05:58.597438Z"},"papermill":{"duration":0.016643,"end_time":"2023-05-02T02:05:58.600607","exception":false,"start_time":"2023-05-02T02:05:58.583964","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define Landmark Indices","metadata":{"papermill":{"duration":0.008382,"end_time":"2023-05-02T02:05:58.617422","exception":false,"start_time":"2023-05-02T02:05:58.609040","status":"completed"},"tags":[]}},{"cell_type":"code","source":"USE_TYPES = ['left_hand', 'pose', 'right_hand']\nSTART_IDX = 468\nLIPS_IDXS0 = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\nLEFT_HAND_IDXS0 = np.arange(468,489)\nRIGHT_HAND_IDXS0 = np.arange(522,543)\nLEFT_POSE_IDXS0 = np.array([502, 504, 506, 508, 510])\nRIGHT_POSE_IDXS0 = np.array([503, 505, 507, 509, 511])\nLANDMARK_IDXS_LEFT_DOMINANT0 = np.concatenate((LIPS_IDXS0, LEFT_HAND_IDXS0, LEFT_POSE_IDXS0))\nLANDMARK_IDXS_RIGHT_DOMINANT0 = np.concatenate((LIPS_IDXS0, RIGHT_HAND_IDXS0, RIGHT_POSE_IDXS0))\nHAND_IDXS0 = np.concatenate((LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0), axis=0)\nN_COLS = LANDMARK_IDXS_LEFT_DOMINANT0.size\nLIPS_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LIPS_IDXS0)).squeeze()\nLEFT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_HAND_IDXS0)).squeeze()\nRIGHT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, RIGHT_HAND_IDXS0)).squeeze()\nHAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, HAND_IDXS0)).squeeze()\nPOSE_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_POSE_IDXS0)).squeeze()\nprint(f'# HAND_IDXS: {len(HAND_IDXS)}, N_COLS: {N_COLS}')","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:05:58.636490Z","iopub.status.busy":"2023-05-02T02:05:58.635550Z","iopub.status.idle":"2023-05-02T02:05:58.649602Z","shell.execute_reply":"2023-05-02T02:05:58.648418Z"},"papermill":{"duration":0.025843,"end_time":"2023-05-02T02:05:58.651673","exception":false,"start_time":"2023-05-02T02:05:58.625830","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LIPS_START = 0\nLEFT_HAND_START = LIPS_IDXS.size\nRIGHT_HAND_START = LEFT_HAND_START + LEFT_HAND_IDXS.size\nPOSE_START = RIGHT_HAND_START + RIGHT_HAND_IDXS.size\nprint(f'LIPS_START: {LIPS_START}, LEFT_HAND_START: {LEFT_HAND_START}, RIGHT_HAND_START: {RIGHT_HAND_START}, POSE_START: {POSE_START}')","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:05:58.670449Z","iopub.status.busy":"2023-05-02T02:05:58.669629Z","iopub.status.idle":"2023-05-02T02:05:58.676311Z","shell.execute_reply":"2023-05-02T02:05:58.675306Z"},"papermill":{"duration":0.01799,"end_time":"2023-05-02T02:05:58.678234","exception":false,"start_time":"2023-05-02T02:05:58.660244","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process Data Tensorflow","metadata":{"papermill":{"duration":0.008465,"end_time":"2023-05-02T02:05:58.695393","exception":false,"start_time":"2023-05-02T02:05:58.686928","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Source: https://www.kaggle.com/competitions/asl-signs/overview/evaluation\nROWS_PER_FRAME = 543  \n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:05:58.715258Z","iopub.status.busy":"2023-05-02T02:05:58.713680Z","iopub.status.idle":"2023-05-02T02:05:58.719765Z","shell.execute_reply":"2023-05-02T02:05:58.718897Z"},"papermill":{"duration":0.017935,"end_time":"2023-05-02T02:05:58.721965","exception":false,"start_time":"2023-05-02T02:05:58.704030","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n    Tensorflow layer to process data in TFLite\n    Data needs to be processed in the model itself, so we can not use Python\n\"\"\" \nclass PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__()\n        normalisation_correction = tf.constant([\n                    # Add 0.50 to left hand (original right hand) and substract 0.50 of right hand (original left hand)\n                    [0] * len(LIPS_IDXS) + [0.50] * len(LEFT_HAND_IDXS) + [0.50] * len(POSE_IDXS),\n                    # Y coordinates stay intact\n                    [0] * len(LANDMARK_IDXS_LEFT_DOMINANT0),\n                    # Z coordinates stay intact\n                    [0] * len(LANDMARK_IDXS_LEFT_DOMINANT0),\n                ],\n                dtype=tf.float32,\n            )\n        self.normalisation_correction = tf.transpose(normalisation_correction, [1,0])\n        \n    def pad_edge(self, t, repeats, side):\n        if side == 'LEFT':\n            return tf.concat((tf.repeat(t[:1], repeats=repeats, axis=0), t), axis=0)\n        elif side == 'RIGHT':\n            return tf.concat((t, tf.repeat(t[-1:], repeats=repeats, axis=0)), axis=0)\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,N_ROWS,N_DIMS], dtype=tf.float32),),\n    )\n    def call(self, data0):\n        \n        # TRUNCATE LONG VIDEOS\n        N_FRAMES0 = tf.shape(data0)[0]\n        data0 = tf.slice(data0, [0,0,0], [tf.math.minimum(INPUT_SIZE * INPUT_SIZE,N_FRAMES0),-1,-1])\n        \n        # Number of Frames in Video\n        N_FRAMES0 = tf.shape(data0)[0]\n        \n        # Find dominant hand by comparing summed absolute coordinates\n        left_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0, 1))\n        right_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0, 1))\n        left_dominant = left_hand_sum >= right_hand_sum\n        \n        # Count non NaN Hand values in each frame for the dominant hand\n        if left_dominant:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0, 1),\n                    axis=[1, 2],\n                )\n        else:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0, 1),\n                    axis=[1, 2],\n                )\n        \n        # Find frames indices with coordinates of dominant hand\n        non_empty_frames_idxs = tf.where(frames_hands_non_nan_sum > 0)\n        non_empty_frames_idxs = tf.squeeze(non_empty_frames_idxs, axis=1)\n        # Filter frames\n        data = tf.gather(data0, non_empty_frames_idxs, axis=0)\n        \n        # Cast Indices in float32 to be compatible with Tensorflow Lite\n        non_empty_frames_idxs = tf.cast(non_empty_frames_idxs, tf.float32)\n        # Normalize to start with 0\n        non_empty_frames_idxs -= tf.reduce_min(non_empty_frames_idxs)\n        \n        # Number of Frames in Filtered Video\n        N_FRAMES = tf.shape(data)[0]\n        \n        # Gather Relevant Landmark Columns\n        if left_dominant:\n            data = tf.gather(data, LANDMARK_IDXS_LEFT_DOMINANT0, axis=1)\n        else:\n            data = tf.gather(data, LANDMARK_IDXS_RIGHT_DOMINANT0, axis=1)\n            data = (\n                    self.normalisation_correction + (\n                        (data - self.normalisation_correction) * tf.where(self.normalisation_correction != 0, -1.0, 1.0))\n                )\n        \n        # Video fits in INPUT_SIZE\n        if N_FRAMES < INPUT_SIZE:\n            # Pad With -1 to indicate padding\n            non_empty_frames_idxs = tf.pad(non_empty_frames_idxs, [[0, INPUT_SIZE-N_FRAMES]], constant_values=-1)\n            # Pad Data With Zeros\n            data = tf.pad(data, [[0, INPUT_SIZE-N_FRAMES], [0,0], [0,0]], constant_values=0)\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            return data, non_empty_frames_idxs\n        # Video needs to be downsampled to INPUT_SIZE\n        else:\n            # Repeat\n            if N_FRAMES < INPUT_SIZE**2:\n                repeats = tf.math.floordiv(INPUT_SIZE * INPUT_SIZE, N_FRAMES0)\n                data = tf.repeat(data, repeats=repeats, axis=0)\n                non_empty_frames_idxs = tf.repeat(non_empty_frames_idxs, repeats=repeats, axis=0)\n\n            # Pad To Multiple Of Input Size\n            pool_size = tf.math.floordiv(len(data), INPUT_SIZE)\n            if tf.math.mod(len(data), INPUT_SIZE) > 0:\n                pool_size += 1\n\n            if pool_size == 1:\n                pad_size = (pool_size * INPUT_SIZE) - len(data)\n            else:\n                pad_size = (pool_size * INPUT_SIZE) % len(data)\n\n            # Pad Start/End with Start/End value\n            pad_left = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(INPUT_SIZE, 2)\n            pad_right = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(INPUT_SIZE, 2)\n            if tf.math.mod(pad_size, 2) > 0:\n                pad_right += 1\n\n            # Pad By Concatenating Left/Right Edge Values\n            data = self.pad_edge(data, pad_left, 'LEFT')\n            data = self.pad_edge(data, pad_right, 'RIGHT')\n\n            # Pad Non Empty Frame Indices\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_left, 'LEFT')\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_right, 'RIGHT')\n\n            # Reshape to Mean Pool\n            data = tf.reshape(data, [INPUT_SIZE, -1, N_COLS, N_DIMS])\n            non_empty_frames_idxs = tf.reshape(non_empty_frames_idxs, [INPUT_SIZE, -1])\n\n            # Mean Pool\n            data = tf.experimental.numpy.nanmean(data, axis=1)\n            non_empty_frames_idxs = tf.experimental.numpy.nanmean(non_empty_frames_idxs, axis=1)\n\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            \n            return data, non_empty_frames_idxs\n    \npreprocess_layer = PreprocessLayer()","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:05:58.740915Z","iopub.status.busy":"2023-05-02T02:05:58.740252Z","iopub.status.idle":"2023-05-02T02:06:01.509263Z","shell.execute_reply":"2023-05-02T02:06:01.508189Z"},"papermill":{"duration":2.781689,"end_time":"2023-05-02T02:06:01.512216","exception":false,"start_time":"2023-05-02T02:05:58.730527","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\n    face: 0:468\n    left_hand: 468:489\n    pose: 489:522\n    right_hand: 522:544   \n\"\"\"\ndef get_data(file_path):\n    # Load Raw Data\n    data = load_relevant_data_subset(file_path)\n    # Process Data Using Tensorflow\n    data = preprocess_layer(data)\n    \n    return data","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:01.531526Z","iopub.status.busy":"2023-05-02T02:06:01.530960Z","iopub.status.idle":"2023-05-02T02:06:01.536840Z","shell.execute_reply":"2023-05-02T02:06:01.535917Z"},"papermill":{"duration":0.017848,"end_time":"2023-05-02T02:06:01.538934","exception":false,"start_time":"2023-05-02T02:06:01.521086","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get the full dataset\ndef preprocess_data():\n    # Create arrays to save data\n    X = np.zeros([N_SAMPLES, INPUT_SIZE, N_COLS, N_DIMS], dtype=np.float32)\n    y = np.zeros([N_SAMPLES], dtype=np.int32)\n    NON_EMPTY_FRAME_IDXS = np.full([N_SAMPLES, INPUT_SIZE], -1, dtype=np.float32)\n\n    # Fill X/y\n    for row_idx, (file_path, sign_ord) in enumerate(tqdm(train[['file_path', 'sign_ord']].values)):\n        # Log message every 5000 samples\n        if row_idx % 5000 == 0:\n            print(f'Generated {row_idx}/{N_SAMPLES}')\n\n        data, non_empty_frame_idxs = get_data(file_path)\n        X[row_idx] = data\n        y[row_idx] = sign_ord\n        NON_EMPTY_FRAME_IDXS[row_idx] = non_empty_frame_idxs\n        # Sanity check, data should not contain NaN values\n        if np.isnan(data).sum() > 0:\n            print(row_idx)\n            return data\n\n    # Save X/y\n    np.save(ROOT_DIR+'/X.npy', X)\n    np.save(ROOT_DIR+'/y.npy', y)\n    np.save(ROOT_DIR+'/NON_EMPTY_FRAME_IDXS.npy', NON_EMPTY_FRAME_IDXS)\n    \n    # Save Validation\n    splitter = GroupShuffleSplit(test_size=0.10, n_splits=2, random_state=SEED)\n    PARTICIPANT_IDS = train['participant_id'].values\n    train_idxs, val_idxs = next(splitter.split(X, y, groups=PARTICIPANT_IDS))\n\n    # Save Train\n    X_train = X[train_idxs]\n    NON_EMPTY_FRAME_IDXS_TRAIN = NON_EMPTY_FRAME_IDXS[train_idxs]\n    y_train = y[train_idxs]\n    np.save(ROOT_DIR+'/X_train.npy', X_train)\n    np.save(ROOT_DIR+'/y_train.npy', y_train)\n    np.save(ROOT_DIR+'/NON_EMPTY_FRAME_IDXS_TRAIN.npy', NON_EMPTY_FRAME_IDXS_TRAIN)\n    # Save Validation\n    X_val = X[val_idxs]\n    NON_EMPTY_FRAME_IDXS_VAL = NON_EMPTY_FRAME_IDXS[val_idxs]\n    y_val = y[val_idxs]\n    np.save(ROOT_DIR+'/X_val.npy', X_val)\n    np.save(ROOT_DIR+'/y_val.npy', y_val)\n    np.save(ROOT_DIR+'/NON_EMPTY_FRAME_IDXS_VAL.npy', NON_EMPTY_FRAME_IDXS_VAL)\n    # Split Statistics\n    print(f'Patient ID Intersection Train/Val: {set(PARTICIPANT_IDS[train_idxs]).intersection(PARTICIPANT_IDS[val_idxs])}')\n    print(f'X_train shape: {X_train.shape}, X_val shape: {X_val.shape}')\n    print(f'y_train shape: {y_train.shape}, y_val shape: {y_val.shape}')","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:01.557635Z","iopub.status.busy":"2023-05-02T02:06:01.557372Z","iopub.status.idle":"2023-05-02T02:06:01.568612Z","shell.execute_reply":"2023-05-02T02:06:01.567576Z"},"papermill":{"duration":0.02332,"end_time":"2023-05-02T02:06:01.570749","exception":false,"start_time":"2023-05-02T02:06:01.547429","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read Training Data\nif (not PREPROCESS_DATA) | DEBUG:\n    train = pd.read_csv('/kaggle/input/asl-signs/train.csv').sample(int(5e3), random_state=SEED)\nelse:\n    train = pd.read_csv('/kaggle/input/asl-signs/train.csv')\n\nN_SAMPLES = len(train)\nprint(f'N_SAMPLES: {N_SAMPLES}')","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:01.589238Z","iopub.status.busy":"2023-05-02T02:06:01.588970Z","iopub.status.idle":"2023-05-02T02:06:01.784035Z","shell.execute_reply":"2023-05-02T02:06:01.782453Z"},"papermill":{"duration":0.20714,"end_time":"2023-05-02T02:06:01.786363","exception":false,"start_time":"2023-05-02T02:06:01.579223","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get complete file path to file\ndef get_file_path(path):\n    return f'/kaggle/input/asl-signs/{path}'\n\ntrain['file_path'] = train['path'].apply(get_file_path)\ntrain['sign_ord'] = train['sign'].astype('category').cat.codes\nORD2SIGN = train[['sign_ord', 'sign']].set_index('sign_ord').squeeze().to_dict()","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:01.806801Z","iopub.status.busy":"2023-05-02T02:06:01.805878Z","iopub.status.idle":"2023-05-02T02:06:01.830605Z","shell.execute_reply":"2023-05-02T02:06:01.829669Z"},"papermill":{"duration":0.037556,"end_time":"2023-05-02T02:06:01.832905","exception":false,"start_time":"2023-05-02T02:06:01.795349","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess All Data From Scratch\nif PREPROCESS_DATA:\n    ROOT_DIR = 'data'\n    os.mkdir(ROOT_DIR)\n    preprocess_data()\nelse:\n    ROOT_DIR = '/kaggle/input/islr-data12'\n    \n# Load Data\nif USE_VAL:\n    # Load Train\n    X_train = np.load(f'{ROOT_DIR}/X_train.npy')\n    y_train = np.load(f'{ROOT_DIR}/y_train.npy')\n    NON_EMPTY_FRAME_IDXS_TRAIN = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS_TRAIN.npy')\n    # Load Val\n    X_val = np.load(f'{ROOT_DIR}/X_val.npy')\n    y_val = np.load(f'{ROOT_DIR}/y_val.npy')\n    NON_EMPTY_FRAME_IDXS_VAL = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS_VAL.npy')\n    # Define validation Data\n    validation_data = ({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, y_val)\nelse:\n    X_train = np.load(f'{ROOT_DIR}/X.npy')\n    y_train = np.load(f'{ROOT_DIR}/y.npy')\n    NON_EMPTY_FRAME_IDXS_TRAIN = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS.npy')\n    validation_data = None\n\n# Train \nprint_shape_dtype([X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN], ['X_train', 'y_train', 'NON_EMPTY_FRAME_IDXS_TRAIN'])\n# Val\nif USE_VAL:\n    print_shape_dtype([X_val, y_val, NON_EMPTY_FRAME_IDXS_VAL], ['X_val', 'y_val', 'NON_EMPTY_FRAME_IDXS_VAL'])\n# Sanity Check\nprint(f'# NaN Values X_train: {np.isnan(X_train).sum()}')","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:01.854135Z","iopub.status.busy":"2023-05-02T02:06:01.852247Z","iopub.status.idle":"2023-05-02T02:06:10.217312Z","shell.execute_reply":"2023-05-02T02:06:10.215813Z"},"papermill":{"duration":8.377855,"end_time":"2023-05-02T02:06:10.219553","exception":false,"start_time":"2023-05-02T02:06:01.841698","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compute Landmark Means and STDs","metadata":{"papermill":{"duration":0.008837,"end_time":"2023-05-02T02:06:10.238027","exception":false,"start_time":"2023-05-02T02:06:10.229190","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def get_lips_mean_std():\n    # LIPS\n    LIPS_MEAN_X = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_MEAN_Y = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_STD_X = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_STD_Y = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n\n    #fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,LIPS_IDXS], [2,3,0,1]).reshape([LIPS_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                LIPS_MEAN_X[col] = v.mean()\n                LIPS_STD_X[col] = v.std()\n            if dim == 1: # Y\n                LIPS_MEAN_Y[col] = v.mean()\n                LIPS_STD_Y[col] = v.std()\n\n            #axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    #for ax, dim_name in zip(axes, DIM_NAMES):\n    #    ax.set_title(f'Lips {dim_name.upper()} Dimension', size=24)\n    #    ax.tick_params(axis='x', labelsize=8)\n    #    ax.grid(axis='y')\n\n    #plt.subplots_adjust(hspace=0.50)\n    #plt.show()\n\n    LIPS_MEAN = np.array([LIPS_MEAN_X, LIPS_MEAN_Y]).T\n    LIPS_STD = np.array([LIPS_STD_X, LIPS_STD_Y]).T\n    \n    return LIPS_MEAN, LIPS_STD\n\nLIPS_MEAN, LIPS_STD = get_lips_mean_std()","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:10.257417Z","iopub.status.busy":"2023-05-02T02:06:10.257102Z","iopub.status.idle":"2023-05-02T02:06:12.857068Z","shell.execute_reply":"2023-05-02T02:06:12.855939Z"},"papermill":{"duration":2.612986,"end_time":"2023-05-02T02:06:12.859818","exception":false,"start_time":"2023-05-02T02:06:10.246832","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_left_right_hand_mean_std():\n    # LEFT HAND\n    LEFT_HANDS_MEAN_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_MEAN_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_STD_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_STD_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n\n    #fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,LEFT_HAND_IDXS], [2,3,0,1]).reshape([LEFT_HAND_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                LEFT_HANDS_MEAN_X[col] = v.mean()\n                LEFT_HANDS_STD_X[col] = v.std()\n            if dim == 1: # Y\n                LEFT_HANDS_MEAN_Y[col] = v.mean()\n                LEFT_HANDS_STD_Y[col] = v.std()\n            # Plot\n            #axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    #for ax, dim_name in zip(axes, DIM_NAMES):\n    #    ax.set_title(f'Hands {dim_name.upper()} Dimension', size=24)\n    #    ax.tick_params(axis='x', labelsize=8)\n    #    ax.grid(axis='y')\n\n    #plt.subplots_adjust(hspace=0.50)\n    #plt.show()\n\n    LEFT_HANDS_MEAN = np.array([LEFT_HANDS_MEAN_X, LEFT_HANDS_MEAN_Y]).T\n    LEFT_HANDS_STD = np.array([LEFT_HANDS_STD_X, LEFT_HANDS_STD_Y]).T\n    \n    return LEFT_HANDS_MEAN, LEFT_HANDS_STD\n\nLEFT_HANDS_MEAN, LEFT_HANDS_STD = get_left_right_hand_mean_std()","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:12.879955Z","iopub.status.busy":"2023-05-02T02:06:12.879626Z","iopub.status.idle":"2023-05-02T02:06:14.195059Z","shell.execute_reply":"2023-05-02T02:06:14.193965Z"},"papermill":{"duration":1.328263,"end_time":"2023-05-02T02:06:14.197802","exception":false,"start_time":"2023-05-02T02:06:12.869539","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_pose_mean_std():\n    # POSE\n    POSE_MEAN_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_MEAN_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_STD_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_STD_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\n\n    #fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,POSE_IDXS], [2,3,0,1]).reshape([POSE_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                POSE_MEAN_X[col] = v.mean()\n                POSE_STD_X[col] = v.std()\n            if dim == 1: # Y\n                POSE_MEAN_Y[col] = v.mean()\n                POSE_STD_Y[col] = v.std()\n\n            #axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    #for ax, dim_name in zip(axes, DIM_NAMES):\n    #    ax.set_title(f'Pose {dim_name.upper()} Dimension', size=24)\n    #    ax.tick_params(axis='x', labelsize=8)\n    #    ax.grid(axis='y')\n\n    #plt.subplots_adjust(hspace=0.50)\n    #plt.show()\n\n    POSE_MEAN = np.array([POSE_MEAN_X, POSE_MEAN_Y]).T\n    POSE_STD = np.array([POSE_STD_X, POSE_STD_Y]).T\n    \n    return POSE_MEAN, POSE_STD\n\nPOSE_MEAN, POSE_STD = get_pose_mean_std()","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:14.217933Z","iopub.status.busy":"2023-05-02T02:06:14.217622Z","iopub.status.idle":"2023-05-02T02:06:14.554954Z","shell.execute_reply":"2023-05-02T02:06:14.553730Z"},"papermill":{"duration":0.351602,"end_time":"2023-05-02T02:06:14.559006","exception":false,"start_time":"2023-05-02T02:06:14.207404","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loader with Augmentation","metadata":{"papermill":{"duration":0.009037,"end_time":"2023-05-02T02:06:14.577810","exception":false,"start_time":"2023-05-02T02:06:14.568773","status":"completed"},"tags":[]}},{"cell_type":"code","source":"TIME_AUG_PROB = 0.25\nFRAME_DROP_PROB = 0.0\n\n# Custom sampler to get a batch containing N times all signs\ndef get_train_batch_all_signs(X, y, NON_EMPTY_FRAME_IDXS, n=BATCH_ALL_SIGNS_N):\n    # Arrays to store batch in\n    X_batch = np.zeros([NUM_CLASSES*n, INPUT_SIZE, N_COLS, N_DIMS], dtype=np.float32)\n    y_batch = np.arange(0, NUM_CLASSES, step=1/n, dtype=np.float32).astype(np.int64)\n    non_empty_frame_idxs_batch = np.zeros([NUM_CLASSES*n, INPUT_SIZE], dtype=np.float32)\n    \n    # Dictionary mapping ordinally encoded sign to corresponding sample indices\n    CLASS2IDXS = {}\n    for i in range(NUM_CLASSES):\n        CLASS2IDXS[i] = np.argwhere(y == i).squeeze().astype(np.int32)\n            \n    while True:\n        # Fill batch arrays\n        for i in range(NUM_CLASSES):\n            idxs = np.random.choice(CLASS2IDXS[i], n)\n            tmp = X[idxs].copy()\n            tmp2 = NON_EMPTY_FRAME_IDXS[idxs].copy()\n            \n            # FRAME DROP AUGMENTATION\n            if np.random.uniform(0,1)<FRAME_DROP_PROB:\n                j = np.random.randint(0,INPUT_SIZE)\n                tmp[0,j,:,:] = 0\n            \n            # TIME SCALE AUGMENTATION\n            if np.random.uniform(0,1)<TIME_AUG_PROB:\n                ct = ( NON_EMPTY_FRAME_IDXS[idxs[0]] != -1 ).sum()\n                if (ct==12):\n                    mask = np.random.choice([True, False],12)\n                    mask[0] = True\n                    mask[-1] = True\n                    c = mask.sum()\n                    tmp[0,:c,:,:] = tmp[0,mask,:,:]\n                    tmp[0,c:,:,:] = 0\n                    tmp2[0,:c] = tmp2[0,mask]\n                    tmp2[0,c:] = -1\n                elif ct>6:\n                    tmp[0,:6,:,:] = tmp[0,::2,:,:]\n                    tmp[0,6:,:,:] = 0\n                    tmp2[0,:6] = tmp2[0,::2]\n                    tmp2[0,6:] = -1\n                elif (ct<=2)&(np.random.uniform(0,1)<0.3):\n                    tmp[0,::6,:,:] = tmp[0,:2,:,:]\n                    tmp[0,1::6,:,:] = tmp[0,:2,:,:]\n                    tmp[0,2::6,:,:] = tmp[0,:2,:,:]\n                    tmp[0,3::6,:,:] = tmp[0,:2,:,:]\n                    tmp[0,4::6,:,:] = tmp[0,:2,:,:]\n                    tmp[0,5::6,:,:] = tmp[0,:2,:,:]\n                    tmp2[0,::6] = tmp2[0,:2]\n                    tmp2[0,1::6] = tmp2[0,:2]\n                    tmp2[0,2::6] = tmp2[0,:2]\n                    tmp2[0,3::6] = tmp2[0,:2]\n                    tmp2[0,4::6] = tmp2[0,:2]\n                    tmp2[0,5::6] = tmp2[0,:2]\n                elif (ct<=3)&(np.random.uniform(0,1)<0.3):\n                    tmp[0,::4,:,:] = tmp[0,:3,:,:]\n                    tmp[0,1::4,:,:] = tmp[0,:3,:,:]\n                    tmp[0,2::4,:,:] = tmp[0,:3,:,:]\n                    tmp[0,3::4,:,:] = tmp[0,:3,:,:]\n                    tmp2[0,::4] = tmp2[0,:3]\n                    tmp2[0,1::4] = tmp2[0,:3]\n                    tmp2[0,2::4] = tmp2[0,:3]\n                    tmp2[0,3::4] = tmp2[0,:3]\n                elif (ct<=4)&(np.random.uniform(0,1)<0.3):\n                    tmp[0,::3,:,:] = tmp[0,:4,:,:]\n                    tmp[0,1::3,:,:] = tmp[0,:4,:,:]\n                    tmp[0,2::3,:,:] = tmp[0,:4,:,:]\n                    tmp2[0,::3] = tmp2[0,:4]\n                    tmp2[0,1::3] = tmp2[0,:4]\n                    tmp2[0,2::3] = tmp2[0,:4]\n                else:\n                    tmp[0,::2,:,:] = tmp[0,:6,:,:]\n                    tmp[0,1::2,:,:] = tmp[0,:6,:,:]\n                    tmp2[0,::2] = tmp2[0,:6]\n                    tmp2[0,1::2] = tmp2[0,:6]\n            \n            X_batch[i*n:(i+1)*n] = tmp\n            non_empty_frame_idxs_batch[i*n:(i+1)*n] = tmp2\n        \n        yield { 'frames': X_batch, 'non_empty_frame_idxs': non_empty_frame_idxs_batch }, y_batch","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:14.597949Z","iopub.status.busy":"2023-05-02T02:06:14.597619Z","iopub.status.idle":"2023-05-02T02:06:14.623311Z","shell.execute_reply":"2023-05-02T02:06:14.622188Z"},"papermill":{"duration":0.038574,"end_time":"2023-05-02T02:06:14.625585","exception":false,"start_time":"2023-05-02T02:06:14.587011","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Config","metadata":{"papermill":{"duration":0.008873,"end_time":"2023-05-02T02:06:14.643576","exception":false,"start_time":"2023-05-02T02:06:14.634703","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Epsilon value for layer normalisation\nLAYER_NORM_EPS = 1e-6\n\n# Dense layer units for landmarks\nLIPS_UNITS = 224 # WAS 384\nHANDS_UNITS = 224 # WAS 384\nPOSE_UNITS = 224 # WAS 384\n# final embedding and transformer embedding size\nUNITS = 376 # WAS 512\n\n# Transformer\nNUM_BLOCKS = 3 # WAS 2\nMLP_RATIO = 3 # WAS 4\n\n# Dropout\nEMBEDDING_DROPOUT = 0.00\nMLP_DROPOUT_RATIO = 0.30 # WAS 0.40\nCLASSIFIER_DROPOUT_RATIO = 0.10\n\n# Initiailizers\nINIT_HE_UNIFORM = tf.keras.initializers.he_uniform\nINIT_GLOROT_UNIFORM = tf.keras.initializers.glorot_uniform\nINIT_ZEROS = tf.keras.initializers.constant(0.0)\n# Activations\nGELU = tf.keras.activations.gelu\n\nprint(f'UNITS: {UNITS}')","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:14.663061Z","iopub.status.busy":"2023-05-02T02:06:14.662714Z","iopub.status.idle":"2023-05-02T02:06:14.669799Z","shell.execute_reply":"2023-05-02T02:06:14.668810Z"},"papermill":{"duration":0.019681,"end_time":"2023-05-02T02:06:14.672248","exception":false,"start_time":"2023-05-02T02:06:14.652567","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define Transformer Model","metadata":{"papermill":{"duration":0.009093,"end_time":"2023-05-02T02:06:14.690611","exception":false,"start_time":"2023-05-02T02:06:14.681518","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# based on: https://stackoverflow.com/\n#questions/67342988/verifying-the-implementation-of-multihead-attention-in-transformer\n# replaced softmax with softmax layer to support masked softmax\ndef scaled_dot_product(q,k,v, softmax, attention_mask):\n    #calculates Q . K(transpose)\n    qkt = tf.matmul(q,k,transpose_b=True)\n    #caculates scaling factor\n    dk = tf.math.sqrt(tf.cast(q.shape[-1],dtype=tf.float32))\n    scaled_qkt = qkt/dk\n    softmax = softmax(scaled_qkt, mask=attention_mask)\n    \n    z = tf.matmul(softmax,v)\n    #shape: (m,Tx,depth), same shape as q,k,v\n    return z\n\nclass MultiHeadAttention(tf.keras.layers.Layer):\n    def __init__(self,d_model,num_of_heads):\n        super(MultiHeadAttention,self).__init__()\n        self.d_model = d_model\n        self.num_of_heads = num_of_heads\n        self.depth = d_model//num_of_heads\n        self.wq = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wk = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wv = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wo = tf.keras.layers.Dense(d_model)\n        self.softmax = tf.keras.layers.Softmax()\n        \n    def call(self,x, attention_mask):\n        \n        multi_attn = []\n        for i in range(self.num_of_heads):\n            Q = self.wq[i](x)\n            K = self.wk[i](x)\n            V = self.wv[i](x)\n            multi_attn.append(scaled_dot_product(Q,K,V, self.softmax, attention_mask))\n            \n        multi_head = tf.concat(multi_attn,axis=-1)\n        multi_head_attention = self.wo(multi_head)\n        return multi_head_attention","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:14.711767Z","iopub.status.busy":"2023-05-02T02:06:14.710139Z","iopub.status.idle":"2023-05-02T02:06:14.721615Z","shell.execute_reply":"2023-05-02T02:06:14.720646Z"},"papermill":{"duration":0.023918,"end_time":"2023-05-02T02:06:14.723851","exception":false,"start_time":"2023-05-02T02:06:14.699933","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Full Transformer\nclass Transformer(tf.keras.Model):\n    def __init__(self, num_blocks):\n        super(Transformer, self).__init__(name='transformer')\n        self.num_blocks = num_blocks\n    \n    def build(self, input_shape):\n        self.ln_1s = []\n        self.mhas = []\n        self.ln_2s = []\n        self.mlps = []\n        # Make Transformer Blocks\n        for i in range(self.num_blocks):\n            # Multi Head Attention\n            self.mhas.append(MultiHeadAttention(UNITS, 8))\n            # Multi Layer Perception\n            self.mlps.append(tf.keras.Sequential([\n                tf.keras.layers.Dense(UNITS * MLP_RATIO, activation=GELU, kernel_initializer=INIT_GLOROT_UNIFORM),\n                tf.keras.layers.Dropout(MLP_DROPOUT_RATIO),\n                tf.keras.layers.Dense(UNITS, kernel_initializer=INIT_HE_UNIFORM),\n            ]))\n        \n    def call(self, x, attention_mask):\n        # Iterate input over transformer blocks\n        for mha, mlp in zip(self.mhas, self.mlps):\n            x = x + mha(x, attention_mask)\n            x = x + mlp(x)\n    \n        return x","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:14.743310Z","iopub.status.busy":"2023-05-02T02:06:14.743004Z","iopub.status.idle":"2023-05-02T02:06:14.750998Z","shell.execute_reply":"2023-05-02T02:06:14.749937Z"},"papermill":{"duration":0.020066,"end_time":"2023-05-02T02:06:14.753071","exception":false,"start_time":"2023-05-02T02:06:14.733005","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LandmarkEmbedding(tf.keras.Model):\n    def __init__(self, units, name):\n        super(LandmarkEmbedding, self).__init__(name=f'{name}_embedding')\n        self.units = units\n        \n    def build(self, input_shape):\n        # Embedding for missing landmark in frame, initizlied with zeros\n        self.empty_embedding = self.add_weight(\n            name=f'{self.name}_empty_embedding',\n            shape=[self.units],\n            initializer=INIT_ZEROS,\n        )\n        # Embedding\n        self.dense = tf.keras.Sequential([\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM),\n            tf.keras.layers.Activation(GELU),\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name=f'{self.name}_dense')\n\n    def call(self, x):\n        return tf.where(\n                # Checks whether landmark is missing in frame\n                tf.reduce_sum(x, axis=2, keepdims=True) == 0,\n                # If so, the empty embedding is used\n                self.empty_embedding,\n                # Otherwise the landmark data is embedded\n                self.dense(x),\n            )","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:14.774026Z","iopub.status.busy":"2023-05-02T02:06:14.772414Z","iopub.status.idle":"2023-05-02T02:06:14.780995Z","shell.execute_reply":"2023-05-02T02:06:14.780006Z"},"papermill":{"duration":0.020767,"end_time":"2023-05-02T02:06:14.783057","exception":false,"start_time":"2023-05-02T02:06:14.762290","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Embedding(tf.keras.Model):\n    def __init__(self):\n        super(Embedding, self).__init__()\n        \n    def get_diffs(self, l):\n        S = l.shape[2]\n        other = tf.expand_dims(l, 3)\n        other = tf.repeat(other, S, axis=3)\n        other = tf.transpose(other, [0,1,3,2])\n        diffs = tf.expand_dims(l, 3) - other\n        diffs = tf.reshape(diffs, [-1, INPUT_SIZE, S*S])\n        return diffs\n\n    def build(self, input_shape):\n        # Positional Embedding, initialized with zeros\n        self.positional_embedding = tf.keras.layers.Embedding(INPUT_SIZE+1, UNITS, embeddings_initializer=INIT_ZEROS)\n        # Embedding layer for Landmarks\n        self.lips_embedding = LandmarkEmbedding(LIPS_UNITS, 'lips')\n        self.left_hand_embedding = LandmarkEmbedding(HANDS_UNITS, 'left_hand')\n        self.pose_embedding = LandmarkEmbedding(POSE_UNITS, 'pose')\n        # Landmark Weights\n        self.landmark_weights = tf.Variable(tf.zeros([3], dtype=tf.float32), name='landmark_weights')\n        # Fully Connected Layers for combined landmarks\n        self.fc = tf.keras.Sequential([\n            tf.keras.layers.Dense(UNITS, name='fully_connected_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM),\n            tf.keras.layers.Activation(GELU),\n            tf.keras.layers.Dense(UNITS, name='fully_connected_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name='fc')\n\n\n    def call(self, lips0, left_hand0, pose0, non_empty_frame_idxs, training=False):\n        # Lips\n        lips_embedding = self.lips_embedding(lips0)\n        # Left Hand\n        left_hand_embedding = self.left_hand_embedding(left_hand0)\n        # Pose\n        pose_embedding = self.pose_embedding(pose0)\n        # Merge Embeddings of all landmarks with mean pooling\n        x = tf.stack((\n            lips_embedding, left_hand_embedding, pose_embedding,\n        ), axis=3)\n        x = x * tf.nn.softmax(self.landmark_weights)\n        x = tf.reduce_sum(x, axis=3)\n        # Fully Connected Layers\n        x = self.fc(x)\n        # Add Positional Embedding\n        max_frame_idxs = tf.clip_by_value(\n                tf.reduce_max(non_empty_frame_idxs, axis=1, keepdims=True),\n                1,\n                np.PINF,\n            )\n        normalised_non_empty_frame_idxs = tf.where(\n            tf.math.equal(non_empty_frame_idxs, -1.0),\n            INPUT_SIZE,\n            tf.cast(\n                non_empty_frame_idxs / max_frame_idxs * INPUT_SIZE,\n                tf.int32,\n            ),\n        )\n        x = x + self.positional_embedding(normalised_non_empty_frame_idxs)\n        \n        return x","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:14.803969Z","iopub.status.busy":"2023-05-02T02:06:14.802359Z","iopub.status.idle":"2023-05-02T02:06:14.816243Z","shell.execute_reply":"2023-05-02T02:06:14.815364Z"},"papermill":{"duration":0.026068,"end_time":"2023-05-02T02:06:14.818335","exception":false,"start_time":"2023-05-02T02:06:14.792267","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Loss Function","metadata":{"papermill":{"duration":0.008964,"end_time":"2023-05-02T02:06:14.836515","exception":false,"start_time":"2023-05-02T02:06:14.827551","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# source:: https://stackoverflow.com/questions/60689185/label-smoothing-for-sparse-categorical-crossentropy\ndef scce_with_ls(y_true, y_pred):\n    # One Hot Encode Sparsely Encoded Target Sign\n    y_true = tf.cast(y_true, tf.int32)\n    y_true = tf.one_hot(y_true, NUM_CLASSES, axis=1)\n    y_true = tf.squeeze(y_true, axis=2)\n    # Categorical Crossentropy with native label smoothing support\n    return tf.keras.losses.categorical_crossentropy(y_true, y_pred, label_smoothing=0.25)","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:14.855929Z","iopub.status.busy":"2023-05-02T02:06:14.855645Z","iopub.status.idle":"2023-05-02T02:06:14.861814Z","shell.execute_reply":"2023-05-02T02:06:14.860916Z"},"papermill":{"duration":0.018452,"end_time":"2023-05-02T02:06:14.864146","exception":false,"start_time":"2023-05-02T02:06:14.845694","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build Model","metadata":{"papermill":{"duration":0.008867,"end_time":"2023-05-02T02:06:14.882549","exception":false,"start_time":"2023-05-02T02:06:14.873682","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def get_model():\n    # Inputs\n    frames = tf.keras.layers.Input([INPUT_SIZE, N_COLS, N_DIMS], dtype=tf.float32, name='frames')\n    non_empty_frame_idxs = tf.keras.layers.Input([INPUT_SIZE], dtype=tf.float32, name='non_empty_frame_idxs')\n    # Padding Mask\n    mask0 = tf.cast(tf.math.not_equal(non_empty_frame_idxs, -1), tf.float32)\n    mask = tf.expand_dims(mask0, axis=2)\n    \n    \"\"\"\n        left_hand: 468:489\n        pose: 489:522\n        right_hand: 522:543\n    \"\"\"\n    x = frames\n    x = tf.slice(x, [0,0,0,0], [-1,INPUT_SIZE, N_COLS, 2])\n    # LIPS\n    lips = tf.slice(x, [0,0,LIPS_START,0], [-1,INPUT_SIZE, 40, 2])\n    lips = tf.where(\n            tf.math.equal(lips, 0.0),\n            0.0,\n            (lips - LIPS_MEAN) / LIPS_STD,\n        )\n    # LEFT HAND\n    left_hand = tf.slice(x, [0,0,40,0], [-1,INPUT_SIZE, 21, 2])\n    left_hand = tf.where(\n            tf.math.equal(left_hand, 0.0),\n            0.0,\n            (left_hand - LEFT_HANDS_MEAN) / LEFT_HANDS_STD,\n        )\n    # POSE\n    pose = tf.slice(x, [0,0,61,0], [-1,INPUT_SIZE, 5, 2])\n    pose = tf.where(\n            tf.math.equal(pose, 0.0),\n            0.0,\n            (pose - POSE_MEAN) / POSE_STD,\n        )\n    \n    # Flatten\n    lips = tf.reshape(lips, [-1, INPUT_SIZE, 40*2])\n    left_hand = tf.reshape(left_hand, [-1, INPUT_SIZE, 21*2])\n    pose = tf.reshape(pose, [-1, INPUT_SIZE, 5*2])\n        \n    # Embedding\n    x = Embedding()(lips, left_hand, pose, non_empty_frame_idxs)\n    \n    # Encoder Transformer Blocks\n    x = Transformer(NUM_BLOCKS)(x, mask)\n    \n    # Pooling\n    x = tf.reduce_sum(x * mask, axis=1) / tf.reduce_sum(mask, axis=1)\n    # Classifier Dropout\n    x = tf.keras.layers.Dropout(CLASSIFIER_DROPOUT_RATIO)(x)\n    # Classification Layer\n    x = tf.keras.layers.Dense(NUM_CLASSES, activation=tf.keras.activations.softmax, kernel_initializer=INIT_GLOROT_UNIFORM)(x)\n    \n    outputs = x\n    \n    # Create Tensorflow Model\n    model = tf.keras.models.Model(inputs=[frames, non_empty_frame_idxs], outputs=outputs)\n    \n    # Sparse Categorical Cross Entropy With Label Smoothing\n    loss = scce_with_ls\n    optimizer = tfa.optimizers.AdamW(learning_rate=1e-3, weight_decay=1e-5, clipnorm=1.0)\n    # TopK Metrics\n    metrics = [\n        tf.keras.metrics.SparseCategoricalAccuracy(name='acc'),\n        tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5, name='top_5_acc'),\n        tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10, name='top_10_acc'),\n    ]\n    \n    model.compile(loss=loss, optimizer=optimizer, metrics=metrics)\n    \n    return model","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:14.902331Z","iopub.status.busy":"2023-05-02T02:06:14.902041Z","iopub.status.idle":"2023-05-02T02:06:14.916992Z","shell.execute_reply":"2023-05-02T02:06:14.916006Z"},"papermill":{"duration":0.027287,"end_time":"2023-05-02T02:06:14.919028","exception":false,"start_time":"2023-05-02T02:06:14.891741","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Learning Rate Scheduler","metadata":{"papermill":{"duration":0.009191,"end_time":"2023-05-02T02:06:14.937436","exception":false,"start_time":"2023-05-02T02:06:14.928245","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def lrfn(current_step, num_warmup_steps, lr_max, num_cycles=0.50, num_training_steps=N_EPOCHS):\n    \n    if current_step < num_warmup_steps:\n        if WARMUP_METHOD == 'log':\n            return lr_max * 0.10 ** (num_warmup_steps - current_step)\n        else:\n            return lr_max * 2 ** -(num_warmup_steps - current_step)\n    else:\n        progress = float(current_step - num_warmup_steps) / float(max(1, num_training_steps - num_warmup_steps))\n\n        return max(0.0, 0.5 * (1.0 + math.cos(math.pi * float(num_cycles) * 2.0 * progress))) * lr_max\n    \nLR_SCHEDULE = [lrfn(step, num_warmup_steps=N_WARMUP_EPOCHS, lr_max=LR_MAX, num_cycles=0.50) for step in range(N_EPOCHS)]\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lambda step: LR_SCHEDULE[step], verbose=1)","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:14.957061Z","iopub.status.busy":"2023-05-02T02:06:14.956760Z","iopub.status.idle":"2023-05-02T02:06:14.965730Z","shell.execute_reply":"2023-05-02T02:06:14.964429Z"},"papermill":{"duration":0.021138,"end_time":"2023-05-02T02:06:14.967797","exception":false,"start_time":"2023-05-02T02:06:14.946659","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Weight Decay Callback","metadata":{"papermill":{"duration":0.009283,"end_time":"2023-05-02T02:06:14.986545","exception":false,"start_time":"2023-05-02T02:06:14.977262","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Custom callback to update weight decay with learning rate\nclass WeightDecayCallback(tf.keras.callbacks.Callback):\n    def __init__(self, wd_ratio=WD_RATIO):\n        self.step_counter = 0\n        self.wd_ratio = wd_ratio\n    \n    def on_epoch_begin(self, epoch, logs=None):\n        model.optimizer.weight_decay = model.optimizer.learning_rate * self.wd_ratio\n        #print(f'learning rate: {model.optimizer.learning_rate.numpy():.2e}, weight decay: {model.optimizer.weight_decay.numpy():.2e}')","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:15.006770Z","iopub.status.busy":"2023-05-02T02:06:15.006447Z","iopub.status.idle":"2023-05-02T02:06:15.012435Z","shell.execute_reply":"2023-05-02T02:06:15.011397Z"},"papermill":{"duration":0.019069,"end_time":"2023-05-02T02:06:15.014881","exception":false,"start_time":"2023-05-02T02:06:14.995812","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train 4 Models","metadata":{"papermill":{"duration":0.008877,"end_time":"2023-05-02T02:06:15.033077","exception":false,"start_time":"2023-05-02T02:06:15.024200","status":"completed"},"tags":[]}},{"cell_type":"code","source":"if TRAIN_MODEL:\n    models = []\n    \n    for k in range(N_MODELS):\n        print('#'*25)\n        print('### Training Model',k,'...')\n        print('#'*25)\n        \n        # Clear all models in GPU\n        VERBOSE = 0; VERBOSE2 = 0\n        if k==0: \n            tf.keras.backend.clear_session()\n            VERBOSE = 2; VERBOSE2 = 1\n\n        # Get new fresh model\n        model = get_model()\n\n        # Sanity Check\n        #model.summary()\n\n        # Actual Training\n        lr_callback = tf.keras.callbacks.LearningRateScheduler(lambda step: LR_SCHEDULE[step], \n                                                               verbose=VERBOSE2)\n        history = model.fit(\n                x=get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN),\n                steps_per_epoch=len(X_train) // (NUM_CLASSES * BATCH_ALL_SIGNS_N),\n                epochs=N_EPOCHS,\n                batch_size=BATCH_SIZE,\n                validation_data=validation_data,\n                callbacks=[\n                    lr_callback,\n                    WeightDecayCallback(),\n                ],\n                verbose = VERBOSE,\n            )\n        \n        # Save Model Weights\n        model.save_weights(f'model{k}.h5')\n        models.append(model)","metadata":{"execution":{"iopub.execute_input":"2023-05-02T02:06:15.052693Z","iopub.status.busy":"2023-05-02T02:06:15.052413Z","iopub.status.idle":"2023-05-02T05:24:49.093392Z","shell.execute_reply":"2023-05-02T05:24:49.092238Z"},"papermill":{"duration":11914.054079,"end_time":"2023-05-02T05:24:49.096339","exception":false,"start_time":"2023-05-02T02:06:15.042260","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Validate 4 Models","metadata":{"papermill":{"duration":0.027436,"end_time":"2023-05-02T05:24:49.152743","exception":false,"start_time":"2023-05-02T05:24:49.125307","status":"completed"},"tags":[]}},{"cell_type":"code","source":"if USE_VAL:\n    preds = []\n    for k in range(len(models)):\n        # Validation Predictions\n        y_val_pred = models[k].predict({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, verbose=2)\n        preds.append(y_val_pred)\n    y_val_pred = np.mean(preds,axis=0).argmax(axis=1)\n    acc = (y_val_pred == y_val).mean()\n    print('Holdout Validation Ensemble ACC =',acc)","metadata":{"execution":{"iopub.execute_input":"2023-05-02T05:24:49.267807Z","iopub.status.busy":"2023-05-02T05:24:49.267446Z","iopub.status.idle":"2023-05-02T05:24:49.273866Z","shell.execute_reply":"2023-05-02T05:24:49.272844Z"},"papermill":{"duration":0.03724,"end_time":"2023-05-02T05:24:49.275928","exception":false,"start_time":"2023-05-02T05:24:49.238688","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Submission\n\nSubmission code loosley based on [this notebook](https://www.kaggle.com/code/dschettler8845/gislr-learn-eda-baseline#baseline) by [Darien Schettler\n](https://www.kaggle.com/dschettler8845)","metadata":{"papermill":{"duration":0.028901,"end_time":"2023-05-02T05:24:49.332560","exception":false,"start_time":"2023-05-02T05:24:49.303659","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# TFLite model for submission\nclass TFLiteModel(tf.Module):\n    def __init__(self, models):\n        super(TFLiteModel, self).__init__()\n\n        # Load the feature generation and main models\n        self.preprocess_layer = preprocess_layer\n        self.models = models\n    \n    @tf.function(input_signature=[tf.TensorSpec(shape=[None, N_ROWS, N_DIMS], dtype=tf.float32, name='inputs')])\n    def __call__(self, inputs):\n        # Preprocess Data\n        x, non_empty_frame_idxs = self.preprocess_layer(inputs)\n        # Add Batch Dimension\n        x = tf.expand_dims(x, axis=0)\n        non_empty_frame_idxs = tf.expand_dims(non_empty_frame_idxs, axis=0)\n        \n        # Make Prediction\n        outputs = []\n        for k in range(len(self.models)):\n            output = self.models[k]({ 'frames': x, 'non_empty_frame_idxs': non_empty_frame_idxs })\n            # Squeeze Output 1x250 -> 250\n            output = tf.squeeze(output, axis=0)\n            outputs.append(output)\n        outputs = tf.math.reduce_mean( outputs, axis=0 )\n\n        # Return a dictionary with the output tensor\n        return {'outputs': outputs}\n\n# Define TF Lite Model\ntflite_keras_model = TFLiteModel(models)\n\n# Sanity Check\ndemo_raw_data = load_relevant_data_subset(train['file_path'].values[5])\nprint(f'demo_raw_data shape: {demo_raw_data.shape}, dtype: {demo_raw_data.dtype}')\ndemo_output = tflite_keras_model(demo_raw_data)[\"outputs\"]\nprint(f'demo_output shape: {demo_output.shape}, dtype: {demo_output.dtype}')\ndemo_prediction = demo_output.numpy().argmax()\nprint(f'demo_prediction: {demo_prediction}, correct: {train.iloc[0][\"sign_ord\"]}')","metadata":{"execution":{"iopub.execute_input":"2023-05-02T05:24:49.432679Z","iopub.status.busy":"2023-05-02T05:24:49.432185Z","iopub.status.idle":"2023-05-02T05:24:56.669526Z","shell.execute_reply":"2023-05-02T05:24:56.667965Z"},"papermill":{"duration":7.290836,"end_time":"2023-05-02T05:24:56.671698","exception":false,"start_time":"2023-05-02T05:24:49.380862","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Model Converter\nkeras_model_converter = tf.lite.TFLiteConverter.from_keras_model(tflite_keras_model)\n\n# QUANTIZE\nQUANTIZE = 'fp16'\nif QUANTIZE in [\"int8\", \"fp16\"]:\n    keras_model_converter.optimizations = [tf.lite.Optimize.DEFAULT]\n    if QUANTIZE == \"fp16\":\n        keras_model_converter.target_spec.supported_types = [tf.float16]\n\n# Convert Model\ntflite_model = keras_model_converter.convert()\n# Write Model\nwith open('/kaggle/working/model.tflite', 'wb') as f:\n    f.write(tflite_model)\n    \n# Zip Model\n!zip submission.zip /kaggle/working/model.tflite","metadata":{"execution":{"iopub.execute_input":"2023-05-02T05:24:56.730809Z","iopub.status.busy":"2023-05-02T05:24:56.730447Z","iopub.status.idle":"2023-05-02T05:28:21.356020Z","shell.execute_reply":"2023-05-02T05:28:21.354659Z"},"papermill":{"duration":204.657974,"end_time":"2023-05-02T05:28:21.358679","exception":false,"start_time":"2023-05-02T05:24:56.700705","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Verify TFLite model can be loaded and used for prediction\n!pip install tflite-runtime\nimport tflite_runtime.interpreter as tflite\n\ninterpreter = tflite.Interpreter(\"/kaggle/working/model.tflite\")\nfound_signatures = list(interpreter.get_signature_list().keys())\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")\n\noutput = prediction_fn(inputs=demo_raw_data)\nsign = output['outputs'].argmax()\n\nprint(\"PRED : \", ORD2SIGN.get(sign), f'[{sign}]')\nprint(\"TRUE : \", train.sign.values[0], f'[{train.sign_ord.values[0]}]')","metadata":{"execution":{"iopub.execute_input":"2023-05-02T05:28:21.417988Z","iopub.status.busy":"2023-05-02T05:28:21.417607Z","iopub.status.idle":"2023-05-02T05:28:32.856718Z","shell.execute_reply":"2023-05-02T05:28:32.854541Z"},"papermill":{"duration":11.471829,"end_time":"2023-05-02T05:28:32.859490","exception":false,"start_time":"2023-05-02T05:28:21.387661","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}