{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"#!pip install tensorflow-addons","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:17.032733Z","iopub.execute_input":"2023-05-06T16:06:17.033165Z","iopub.status.idle":"2023-05-06T16:06:17.061793Z","shell.execute_reply.started":"2023-05-06T16:06:17.033129Z","shell.execute_reply":"2023-05-06T16:06:17.060570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    aggregation_data_path = \"../input/isolated-sign-language-aggregation-dataset/\"\n    data_path = \"../input/asl-signs/\"\n    pred_model = '../input/sign-language-prediction-model/'\n    make_featuregen = False\n    quick_experiment = False\n    kfold_training = False\n    DROP_Z = False\n    is_training = False\n    make_aggregation_dataset = False\n    num_classes = 250\n    rows_per_frame = 543 ","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:17.064390Z","iopub.execute_input":"2023-05-06T16:06:17.064751Z","iopub.status.idle":"2023-05-06T16:06:17.076225Z","shell.execute_reply.started":"2023-05-06T16:06:17.064715Z","shell.execute_reply":"2023-05-06T16:06:17.075049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Packages","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tqdm.notebook import tqdm\nimport json\nimport os\nimport gc\nimport tensorflow_addons as tfa\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom types import SimpleNamespace\nfrom types import SimpleNamespace\nfrom pathlib import Path\nimport math\nimport scipy\nimport matplotlib.pyplot as plt\n\n\n#LANDMARK_IDX = [0,9,11,13,14,17,117,118,119,199,346,347,348] + list(range(468,543))","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:17.077509Z","iopub.execute_input":"2023-05-06T16:06:17.077832Z","iopub.status.idle":"2023-05-06T16:06:27.939166Z","shell.execute_reply.started":"2023-05-06T16:06:17.077784Z","shell.execute_reply":"2023-05-06T16:06:27.937742Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utilities","metadata":{}},{"cell_type":"code","source":"def load_relevant_data_subset_with_imputation(pq_path):\n    data_columns = ['landmark_index','x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    data['landmark_index'] = list(range(0,543))*int(len(data) / CFG.rows_per_frame)\n    data.replace(np.nan, 0, inplace=True)\n    data.drop(data[~data['landmark_index'].isin(LANDMARK_IDX)].index, inplace=True)\n    data = data.drop('landmark_index', axis=1)\n    n_frames = int(len(data) / len(LANDMARK_IDX))\n    data = data.values.reshape(n_frames, len(LANDMARK_IDX), 3)\n    return data.astype(np.float32)\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / CFG.rows_per_frame)\n    data = data.values.reshape(n_frames, CFG.rows_per_frame, len(data_columns))\n    return data.astype(np.float32)\n\ndef read_dict(file_path):\n    path = os.path.expanduser(file_path)\n    with open(path, \"r\") as f:\n        dic = json.load(f)\n    return dic","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:27.942043Z","iopub.execute_input":"2023-05-06T16:06:27.943096Z","iopub.status.idle":"2023-05-06T16:06:27.955906Z","shell.execute_reply.started":"2023-05-06T16:06:27.943037Z","shell.execute_reply":"2023-05-06T16:06:27.954591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(f\"{CFG.aggregation_data_path}train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:27.960432Z","iopub.execute_input":"2023-05-06T16:06:27.960817Z","iopub.status.idle":"2023-05-06T16:06:28.272772Z","shell.execute_reply.started":"2023-05-06T16:06:27.960778Z","shell.execute_reply":"2023-05-06T16:06:28.271185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 21 participants. Each of them create about 3000 to 5000 training records.","metadata":{}},{"cell_type":"code","source":"train.participant_id.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:28.274191Z","iopub.execute_input":"2023-05-06T16:06:28.274566Z","iopub.status.idle":"2023-05-06T16:06:28.287927Z","shell.execute_reply.started":"2023-05-06T16:06:28.274532Z","shell.execute_reply":"2023-05-06T16:06:28.286924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.participant_id.value_counts().plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:28.289548Z","iopub.execute_input":"2023-05-06T16:06:28.290159Z","iopub.status.idle":"2023-05-06T16:06:28.936071Z","shell.execute_reply.started":"2023-05-06T16:06:28.290113Z","shell.execute_reply":"2023-05-06T16:06:28.934615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 94477 training samples in total.","metadata":{}},{"cell_type":"markdown","source":"There are 250 kinds of sign languages that we need to make prediction on. Each kind of sign languages contains about 300 to 400 samples.","metadata":{}},{"cell_type":"code","source":"label_index = read_dict(f\"{CFG.data_path}sign_to_prediction_index_map.json\")\nindex_label = dict([(label_index[key], key) for key in label_index])\nprint(label_index)\ntrain[\"label\"] = train[\"sign\"].map(lambda sign: label_index[sign])\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:28.937686Z","iopub.execute_input":"2023-05-06T16:06:28.938061Z","iopub.status.idle":"2023-05-06T16:06:29.002892Z","shell.execute_reply.started":"2023-05-06T16:06:28.938016Z","shell.execute_reply":"2023-05-06T16:06:29.001594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"sign\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:29.004491Z","iopub.execute_input":"2023-05-06T16:06:29.004986Z","iopub.status.idle":"2023-05-06T16:06:29.021677Z","shell.execute_reply.started":"2023-05-06T16:06:29.004933Z","shell.execute_reply":"2023-05-06T16:06:29.020275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here are descriptive statistics for number of frames.","metadata":{}},{"cell_type":"code","source":"train.num_frames.describe()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:29.023140Z","iopub.execute_input":"2023-05-06T16:06:29.024229Z","iopub.status.idle":"2023-05-06T16:06:29.041838Z","shell.execute_reply.started":"2023-05-06T16:06:29.024183Z","shell.execute_reply":"2023-05-06T16:06:29.040834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# if CFG.make_aggregation_dataset:\n#     xs = []\n#     ys = []\n#     num_frames = np.zeros(len(train))\n#     for i in tqdm(range(len(train))):\n#         path = f\"{CFG.data_path}{train.iloc[i].path}\"\n#         data = load_relevant_data_subset_with_imputation(path)\n#         ## Mean Aggregation\n#         xs.append(np.mean(data, axis=0))\n#         ys.append(train.iloc[i].label)\n#         num_frames[i] = data.shape[0]\n#         if CFG.quick_experiment and i == 4999:\n#             break\n#     ## Save number of frames of each training sample for data analysis\n#     train[\"num_frames\"] = num_frames\n#     X = np.array(xs)\n#     y = np.array(ys)\n#     print(train[\"num_frames\"].describe())\n#     train.to_csv(\"train.csv\", index=False)\n# else:\n#     X = np.load(\"X_dropped.npy\")\n#     y = np.load(\"y_dropped.npy\")\n# print(X.shape, y.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:29.043817Z","iopub.execute_input":"2023-05-06T16:06:29.044687Z","iopub.status.idle":"2023-05-06T16:06:29.049255Z","shell.execute_reply.started":"2023-05-06T16:06:29.044632Z","shell.execute_reply":"2023-05-06T16:06:29.048361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Transformer","metadata":{}},{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"cfg = SimpleNamespace()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:29.050918Z","iopub.execute_input":"2023-05-06T16:06:29.051715Z","iopub.status.idle":"2023-05-06T16:06:29.066993Z","shell.execute_reply.started":"2023-05-06T16:06:29.051634Z","shell.execute_reply":"2023-05-06T16:06:29.065592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iskaggle = os.environ.get('KAGGLE_KERNEL_RUN_TYPE', '')","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:29.068890Z","iopub.execute_input":"2023-05-06T16:06:29.069629Z","iopub.status.idle":"2023-05-06T16:06:29.079102Z","shell.execute_reply.started":"2023-05-06T16:06:29.069587Z","shell.execute_reply":"2023-05-06T16:06:29.078085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR         = Path('../data/') if not iskaggle else Path('/kaggle/input/asl-signs/')\nTRAIN_CSV_PATH   = DATA_DIR/'train.csv'\nLANDMARK_DIR     = DATA_DIR/'train_landmark_files'\nLABEL_MAP_PATH   = DATA_DIR/'sign_to_prediction_index_map.json'","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:29.084754Z","iopub.execute_input":"2023-05-06T16:06:29.085189Z","iopub.status.idle":"2023-05-06T16:06:29.092242Z","shell.execute_reply.started":"2023-05-06T16:06:29.085149Z","shell.execute_reply":"2023-05-06T16:06:29.090913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg.PREPROCESS_DATA = False\ncfg.TRAIN_MODEL = False\ncfg.N_ROWS = 543\ncfg.N_DIMS = 3\ncfg.DIM_NAMES = ['x', 'y', 'z']\ncfg.SEED = 42\ncfg.NUM_CLASSES = 250\ncfg.IS_INTERACTIVE = True\ncfg.VERBOSE = 2\ncfg.INPUT_SIZE = 32\ncfg.BATCH_ALL_SIGNS_N = 4\ncfg.BATCH_SIZE = 256\ncfg.N_EPOCHS = 100\ncfg.LR_MAX = 1e-3\ncfg.N_WARMUP_EPOCHS = 0\ncfg.WD_RATIO = 0.05\ncfg.MASK_VAL = 4237","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:29.093921Z","iopub.execute_input":"2023-05-06T16:06:29.094333Z","iopub.status.idle":"2023-05-06T16:06:29.104467Z","shell.execute_reply.started":"2023-05-06T16:06:29.094293Z","shell.execute_reply":"2023-05-06T16:06:29.103363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(TRAIN_CSV_PATH)\nN_SAMPLES = len(train)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:29.106060Z","iopub.execute_input":"2023-05-06T16:06:29.107402Z","iopub.status.idle":"2023-05-06T16:06:29.326635Z","shell.execute_reply.started":"2023-05-06T16:06:29.107353Z","shell.execute_reply":"2023-05-06T16:06:29.325353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get complete file path to file\ndef get_file_path(path):\n    return f'/kaggle/input/asl-signs/{path}'\n\n!ln -s {LANDMARK_DIR} ./train_landmark_files\ntrain['file_path'] = train['path'].values\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:29.328098Z","iopub.execute_input":"2023-05-06T16:06:29.328502Z","iopub.status.idle":"2023-05-06T16:06:30.446433Z","shell.execute_reply.started":"2023-05-06T16:06:29.328463Z","shell.execute_reply":"2023-05-06T16:06:30.444712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['sign_ord'] = train['sign'].astype('category').cat.codes\nSIGN2ORD = train[['sign', 'sign_ord']].set_index('sign').squeeze().to_dict()\nORD2SIGN = train[['sign_ord', 'sign']].set_index('sign_ord').squeeze().to_dict()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:30.448840Z","iopub.execute_input":"2023-05-06T16:06:30.449572Z","iopub.status.idle":"2023-05-06T16:06:30.642301Z","shell.execute_reply.started":"2023-05-06T16:06:30.449519Z","shell.execute_reply":"2023-05-06T16:06:30.641355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_PARTICIPANTS = train.participant_id.nunique()\nsgkf = StratifiedGroupKFold(n_splits=7, shuffle=True, random_state=43)\ntrain['fold'] = -1\nfor i, (train_idx, val_idx) in enumerate(sgkf.split(train.index, train.sign, train.participant_id)):\n    train.loc[val_idx, 'fold'] = i\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:30.643846Z","iopub.execute_input":"2023-05-06T16:06:30.644926Z","iopub.status.idle":"2023-05-06T16:06:31.018936Z","shell.execute_reply.started":"2023-05-06T16:06:30.644880Z","shell.execute_reply":"2023-05-06T16:06:31.017623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create indexes using fold `0` for now\ntrain_idxs = train.query(\"fold!=0\").index.values\nval_idxs = train.query(\"fold==0\").index.values","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:31.020770Z","iopub.execute_input":"2023-05-06T16:06:31.021573Z","iopub.status.idle":"2023-05-06T16:06:31.056167Z","shell.execute_reply.started":"2023-05-06T16:06:31.021507Z","shell.execute_reply":"2023-05-06T16:06:31.054943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_idxs), len(val_idxs)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:31.058317Z","iopub.execute_input":"2023-05-06T16:06:31.058812Z","iopub.status.idle":"2023-05-06T16:06:31.067644Z","shell.execute_reply.started":"2023-05-06T16:06:31.058761Z","shell.execute_reply":"2023-05-06T16:06:31.066155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"# landmark indices in original data\nLIPS_IDXS0 = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\nLEFT_HAND_IDXS0  = np.arange(468,489)\nRIGHT_HAND_IDXS0 = np.arange(522,543)\nPOSE_IDXS0       = np.arange(502, 512)\nLANDMARK_IDXS0   = np.concatenate((LIPS_IDXS0, LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0, POSE_IDXS0))\nHAND_IDXS0       = np.concatenate((LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0), axis=0)\nN_COLS           = LANDMARK_IDXS0.size\nN_COLS, LANDMARK_IDXS0","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:31.069506Z","iopub.execute_input":"2023-05-06T16:06:31.070023Z","iopub.status.idle":"2023-05-06T16:06:31.083974Z","shell.execute_reply.started":"2023-05-06T16:06:31.069971Z","shell.execute_reply":"2023-05-06T16:06:31.082491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Landmark indices in processed data\nLIPS_IDXS       = np.argwhere(np.isin(LANDMARK_IDXS0, LIPS_IDXS0)).squeeze()\nLEFT_HAND_IDXS  = np.argwhere(np.isin(LANDMARK_IDXS0, LEFT_HAND_IDXS0)).squeeze()\nRIGHT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS0, RIGHT_HAND_IDXS0)).squeeze()\nHAND_IDXS       = np.argwhere(np.isin(LANDMARK_IDXS0, HAND_IDXS0)).squeeze()\nPOSE_IDXS       = np.argwhere(np.isin(LANDMARK_IDXS0, POSE_IDXS0)).squeeze()\nprint(HAND_IDXS.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:31.085957Z","iopub.execute_input":"2023-05-06T16:06:31.086491Z","iopub.status.idle":"2023-05-06T16:06:31.098626Z","shell.execute_reply.started":"2023-05-06T16:06:31.086438Z","shell.execute_reply":"2023-05-06T16:06:31.097372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__()\n        \n    def pad_edge(self, t, repeats, side):\n        if side == 'LEFT':\n            return tf.concat((tf.repeat(t[:1], repeats=repeats, axis=0), t), axis=0)\n        elif side == 'RIGHT':\n            return tf.concat((t, tf.repeat(t[-1:], repeats=repeats, axis=0)), axis=0)\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,cfg.N_ROWS,cfg.N_DIMS], dtype=tf.float32),),\n    )\n    def call(self, data0):\n        # Number of Frames in Video\n        N_FRAMES0 = tf.shape(data0)[0]\n        \n        # Keep only non-empty frames in data\n        frames_hands_nansum = tf.experimental.numpy.nanmean(tf.gather(data0, HAND_IDXS0, axis=1), axis=[1,2])\n        non_empty_frames_idxs = tf.where(frames_hands_nansum > 0)\n        non_empty_frames_idxs = tf.squeeze(non_empty_frames_idxs, axis=1)\n        data = tf.gather(data0, non_empty_frames_idxs, axis=0)\n        \n        non_empty_frames_idxs = tf.cast(non_empty_frames_idxs, tf.float32) \n        \n        # Number of non-empty frames\n        N_FRAMES = tf.shape(data)[0]\n        data = tf.gather(data, LANDMARK_IDXS0, axis=1)\n        \n        if N_FRAMES < cfg.INPUT_SIZE:\n            # Video fits in cfg.INPUT_SIZE\n            non_empty_frames_idxs = tf.pad(non_empty_frames_idxs, [[0, cfg.INPUT_SIZE-N_FRAMES]], constant_values=-1)\n            data = tf.pad(data, [[0, cfg.INPUT_SIZE-N_FRAMES], [0,0], [0,0]], constant_values=0)\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            return data, non_empty_frames_idxs\n        else:\n            # Video needs to be downsampled to cfg.INPUT_SIZE\n            if N_FRAMES < cfg.INPUT_SIZE**2:\n                repeats = tf.math.floordiv(cfg.INPUT_SIZE * cfg.INPUT_SIZE, N_FRAMES0)\n                data = tf.repeat(data, repeats=repeats, axis=0)\n                non_empty_frames_idxs = tf.repeat(non_empty_frames_idxs, repeats=repeats, axis=0)\n            \n            # Pad To Multiple Of Input Size\n            pool_size = tf.math.floordiv(len(data), cfg.INPUT_SIZE)\n            if tf.math.mod(len(data), cfg.INPUT_SIZE) > 0:\n                pool_size += 1\n            if pool_size == 1:\n                pad_size = (pool_size * cfg.INPUT_SIZE) - len(data)\n            else:\n                pad_size = (pool_size * cfg.INPUT_SIZE) % len(data)\n\n            # Pad Start/End with Start/End value\n            pad_left = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(cfg.INPUT_SIZE, 2)\n            pad_right = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(cfg.INPUT_SIZE, 2)\n            if tf.math.mod(pad_size, 2) > 0:\n                pad_right += 1\n\n            # Pad By Concatenating Left/Right Edge Values\n            data = self.pad_edge(data, pad_left, 'LEFT')\n            data = self.pad_edge(data, pad_right, 'RIGHT')\n\n            # Pad Non Empty Frame Indices\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_left, 'LEFT')\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_right, 'RIGHT')\n\n            # Reshape to Mean Pool\n            data = tf.reshape(data, [cfg.INPUT_SIZE, -1, N_COLS, cfg.N_DIMS])\n            non_empty_frames_idxs = tf.reshape(non_empty_frames_idxs, [cfg.INPUT_SIZE, -1])\n\n            # Mean Pool\n            data = tf.experimental.numpy.nanmean(data, axis=1)\n            non_empty_frames_idxs = tf.experimental.numpy.nanmean(non_empty_frames_idxs, axis=1)\n\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            \n            return data, non_empty_frames_idxs\n    \npreprocess_layer = PreprocessLayer()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:31.100631Z","iopub.execute_input":"2023-05-06T16:06:31.100989Z","iopub.status.idle":"2023-05-06T16:06:31.139025Z","shell.execute_reply.started":"2023-05-06T16:06:31.100955Z","shell.execute_reply":"2023-05-06T16:06:31.137818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = load_relevant_data_subset(train.path[0])\nsample.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:31.140493Z","iopub.execute_input":"2023-05-06T16:06:31.141157Z","iopub.status.idle":"2023-05-06T16:06:31.279215Z","shell.execute_reply.started":"2023-05-06T16:06:31.141121Z","shell.execute_reply":"2023-05-06T16:06:31.278166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data, non_empty_frames_idxs = preprocess_layer(sample)\ndata.shape, non_empty_frames_idxs","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:31.280543Z","iopub.execute_input":"2023-05-06T16:06:31.281169Z","iopub.status.idle":"2023-05-06T16:06:32.576762Z","shell.execute_reply.started":"2023-05-06T16:06:31.281129Z","shell.execute_reply":"2023-05-06T16:06:32.575361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# free up RAM, delete variables as we go\ndel data; del non_empty_frames_idxs","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:32.580546Z","iopub.execute_input":"2023-05-06T16:06:32.581160Z","iopub.status.idle":"2023-05-06T16:06:32.587145Z","shell.execute_reply.started":"2023-05-06T16:06:32.581107Z","shell.execute_reply":"2023-05-06T16:06:32.585390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Dataset","metadata":{}},{"cell_type":"code","source":"LIPS_START = 0\nLEFT_HAND_START = LIPS_IDXS.size\nRIGHT_HAND_START = LEFT_HAND_START + LEFT_HAND_IDXS.size\nPOSE_START = RIGHT_HAND_START + RIGHT_HAND_IDXS.size","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:32.588731Z","iopub.execute_input":"2023-05-06T16:06:32.589144Z","iopub.status.idle":"2023-05-06T16:06:32.601030Z","shell.execute_reply.started":"2023-05-06T16:06:32.589102Z","shell.execute_reply":"2023-05-06T16:06:32.599687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_data(file_path):\n    data = load_relevant_data_subset(file_path)\n    data = preprocess_layer(data)\n    return data","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:32.602545Z","iopub.execute_input":"2023-05-06T16:06:32.603491Z","iopub.status.idle":"2023-05-06T16:06:32.618497Z","shell.execute_reply.started":"2023-05-06T16:06:32.603449Z","shell.execute_reply":"2023-05-06T16:06:32.616845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_x_y():\n    # Create arrays to save data\n    X = np.zeros([N_SAMPLES, cfg.INPUT_SIZE, N_COLS, cfg.N_DIMS], dtype=np.float32)\n    y = np.zeros([N_SAMPLES], dtype=np.int32)\n    NON_EMPTY_FRAME_IDXS = np.full([N_SAMPLES, cfg.INPUT_SIZE], -1, dtype=np.float32)\n    print(NON_EMPTY_FRAME_IDXS)\n\n    for row_idx, (file_path, sign_ord) in enumerate(tqdm(train[['file_path', 'sign_ord']].values)):\n        if row_idx % 5000 == 0:\n            print(f'Generated {row_idx}/{N_SAMPLES}')\n\n        data, non_empty_frame_idxs = get_data(file_path)\n        X[row_idx] = data\n        y[row_idx] = sign_ord\n        NON_EMPTY_FRAME_IDXS[row_idx] = non_empty_frame_idxs\n        if np.isnan(data).sum() > 0: return data\n\n    # Save X/y\n    np.save('X.npy', X)\n    np.save('y.npy', y)\n    np.save('NON_EMPTY_FRAME_IDXS.npy', NON_EMPTY_FRAME_IDXS)\n    return X, y, NON_EMPTY_FRAME_IDXS","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:32.620144Z","iopub.execute_input":"2023-05-06T16:06:32.620667Z","iopub.status.idle":"2023-05-06T16:06:32.631753Z","shell.execute_reply.started":"2023-05-06T16:06:32.620606Z","shell.execute_reply":"2023-05-06T16:06:32.630589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if cfg.PREPROCESS_DATA:\n    X, y, NON_EMPTY_FRAME_IDXS = get_x_y()\nelse:\n    X = np.load('/kaggle/input/islr-feature-gen-dataset/X.npy')\n    y = np.load('/kaggle/input/islr-feature-gen-dataset/y.npy')\n    NON_EMPTY_FRAME_IDXS = np.load('/kaggle/input/islr-feature-gen-dataset/NON_EMPTY_FRAME_IDXS.npy')","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:32.633475Z","iopub.execute_input":"2023-05-06T16:06:32.633861Z","iopub.status.idle":"2023-05-06T16:06:54.475800Z","shell.execute_reply.started":"2023-05-06T16:06:32.633824Z","shell.execute_reply":"2023-05-06T16:06:54.474358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape, y.shape, NON_EMPTY_FRAME_IDXS.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:54.477727Z","iopub.execute_input":"2023-05-06T16:06:54.478847Z","iopub.status.idle":"2023-05-06T16:06:54.486861Z","shell.execute_reply.started":"2023-05-06T16:06:54.478773Z","shell.execute_reply":"2023-05-06T16:06:54.485716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Statistics","metadata":{}},{"cell_type":"code","source":"# LIPS\nLIPS_MEAN_X  = np.zeros([LIPS_IDXS.size], dtype=np.float32)\nLIPS_MEAN_Y  = np.zeros([LIPS_IDXS.size], dtype=np.float32)\nLIPS_STD_X   = np.zeros([LIPS_IDXS.size], dtype=np.float32)\nLIPS_STD_Y   = np.zeros([LIPS_IDXS.size], dtype=np.float32)\nfor col, ll in enumerate(tqdm( np.transpose(X[:,:,LIPS_IDXS], [2,3,0,1]).reshape([LIPS_IDXS.size, cfg.N_DIMS, -1]) )):\n    for dim, l in enumerate(ll):\n        #print(dim, len(l))\n        v = l[np.nonzero(l)]\n        if dim == 0: # X\n            LIPS_MEAN_X[col] = v.mean()\n            LIPS_STD_X[col] = v.std()\n        if dim == 1: # Y\n            LIPS_MEAN_Y[col] = v.mean()\n            LIPS_STD_Y[col] = v.std()\n        \nLIPS_MEAN = np.array([LIPS_MEAN_X, LIPS_MEAN_Y]).T\nLIPS_STD = np.array([LIPS_STD_X, LIPS_STD_Y]).T","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:06:54.488367Z","iopub.execute_input":"2023-05-06T16:06:54.488820Z","iopub.status.idle":"2023-05-06T16:07:05.330142Z","shell.execute_reply.started":"2023-05-06T16:06:54.488787Z","shell.execute_reply":"2023-05-06T16:07:05.328700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LEFT HAND\nLEFT_HANDS_MEAN_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\nLEFT_HANDS_MEAN_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\nLEFT_HANDS_STD_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\nLEFT_HANDS_STD_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n# RIGHT HAND\nRIGHT_HANDS_MEAN_X = np.zeros([RIGHT_HAND_IDXS.size], dtype=np.float32)\nRIGHT_HANDS_MEAN_Y = np.zeros([RIGHT_HAND_IDXS.size], dtype=np.float32)\nRIGHT_HANDS_STD_X = np.zeros([RIGHT_HAND_IDXS.size], dtype=np.float32)\nRIGHT_HANDS_STD_Y = np.zeros([RIGHT_HAND_IDXS.size], dtype=np.float32)\n\nfor col, ll in enumerate(tqdm( np.transpose(X[:,:,HAND_IDXS], [2,3,0,1]).reshape([HAND_IDXS.size, cfg.N_DIMS, -1]))):\n    for dim, l in enumerate(ll):\n        v = l[np.nonzero(l)]\n        if dim == 0: # X\n            if col < RIGHT_HAND_IDXS.size: # LEFT HAND\n                LEFT_HANDS_MEAN_X[col] = v.mean()\n                LEFT_HANDS_STD_X[col] = v.std()\n            else:\n                RIGHT_HANDS_MEAN_X[col - LEFT_HAND_IDXS.size] = v.mean()\n                RIGHT_HANDS_STD_X[col - LEFT_HAND_IDXS.size] = v.std()\n        if dim == 1: # Y\n            if col < RIGHT_HAND_IDXS.size: # LEFT HAND\n                LEFT_HANDS_MEAN_Y[col] = v.mean()\n                LEFT_HANDS_STD_Y[col] = v.std()\n            else: # RIGHT HAND\n                RIGHT_HANDS_MEAN_Y[col - LEFT_HAND_IDXS.size] = v.mean()\n                RIGHT_HANDS_STD_Y[col - LEFT_HAND_IDXS.size] = v.std()\n        \nLEFT_HANDS_MEAN = np.array([LEFT_HANDS_MEAN_X, LEFT_HANDS_MEAN_Y]).T\nLEFT_HANDS_STD = np.array([LEFT_HANDS_STD_X, LEFT_HANDS_STD_Y]).T\nRIGHT_HANDS_MEAN = np.array([RIGHT_HANDS_MEAN_X, RIGHT_HANDS_MEAN_Y]).T\nRIGHT_HANDS_STD = np.array([RIGHT_HANDS_STD_X, RIGHT_HANDS_STD_Y]).T","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:05.331783Z","iopub.execute_input":"2023-05-06T16:07:05.332209Z","iopub.status.idle":"2023-05-06T16:07:16.039088Z","shell.execute_reply.started":"2023-05-06T16:07:05.332168Z","shell.execute_reply":"2023-05-06T16:07:16.037659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# POSE\nPOSE_MEAN_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\nPOSE_MEAN_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\nPOSE_STD_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\nPOSE_STD_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\n\nfor col, ll in enumerate(tqdm( np.transpose(X[:,:,POSE_IDXS], [2,3,0,1]).reshape([POSE_IDXS.size, cfg.N_DIMS, -1]) )):\n    for dim, l in enumerate(ll):\n        v = l[np.nonzero(l)]\n        if dim == 0: # X\n            POSE_MEAN_X[col] = v.mean()\n            POSE_STD_X[col] = v.std()\n        if dim == 1: # Y\n            POSE_MEAN_Y[col] = v.mean()\n            POSE_STD_Y[col] = v.std()\n        \nPOSE_MEAN = np.array([POSE_MEAN_X, POSE_MEAN_Y]).T\nPOSE_STD = np.array([POSE_STD_X, POSE_STD_Y]).T","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:16.041435Z","iopub.execute_input":"2023-05-06T16:07:16.041935Z","iopub.status.idle":"2023-05-06T16:07:18.683598Z","shell.execute_reply.started":"2023-05-06T16:07:16.041880Z","shell.execute_reply":"2023-05-06T16:07:18.682372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Custom sampler to get a batch containing N times all signs\ndef get_train_batch_all_signs(X, y, NON_EMPTY_FRAME_IDXS, n=cfg.BATCH_ALL_SIGNS_N):\n    # Arrays to store batch in\n    X_batch = np.zeros([cfg.NUM_CLASSES*n, cfg.INPUT_SIZE, N_COLS, cfg.N_DIMS], dtype=np.float32)\n    y_batch = np.arange(0, cfg.NUM_CLASSES, step=1/n, dtype=np.float32).astype(np.int64)\n    non_empty_frame_idxs_batch = np.zeros([cfg.NUM_CLASSES*n, cfg.INPUT_SIZE], dtype=np.float32)\n    \n    # Dictionary mapping ordinally encoded sign to corresponding sample indices\n    CLASS2IDXS = {}\n    for i in range(cfg.NUM_CLASSES):\n        CLASS2IDXS[i] = np.argwhere(y == i).squeeze().astype(np.int32)\n            \n    while True:\n        # Fill batch arrays\n        for i in range(cfg.NUM_CLASSES):\n            idxs = np.random.choice(CLASS2IDXS[i], n)\n            X_batch[i*n:(i+1)*n] = X[idxs]\n            non_empty_frame_idxs_batch[i*n:(i+1)*n] = NON_EMPTY_FRAME_IDXS[idxs]\n        \n        yield { 'frames': X_batch, 'non_empty_frame_idxs': non_empty_frame_idxs_batch }, y_batch","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:18.685207Z","iopub.execute_input":"2023-05-06T16:07:18.685698Z","iopub.status.idle":"2023-05-06T16:07:18.698952Z","shell.execute_reply.started":"2023-05-06T16:07:18.685648Z","shell.execute_reply":"2023-05-06T16:07:18.697362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dummy_dataset = get_train_batch_all_signs(X, y, NON_EMPTY_FRAME_IDXS)\nX_batch, y_batch = next(dummy_dataset)\n#print(X_batch)\nX_batch.keys(), X_batch['frames'].shape","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:18.701383Z","iopub.execute_input":"2023-05-06T16:07:18.701809Z","iopub.status.idle":"2023-05-06T16:07:18.775722Z","shell.execute_reply.started":"2023-05-06T16:07:18.701770Z","shell.execute_reply":"2023-05-06T16:07:18.774329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model config","metadata":{}},{"cell_type":"code","source":"# Epsilon value for layer normalisation\nLAYER_NORM_EPS = 1e-6\n\n# Dense layer units for landmarks\nLIPS_UNITS = 384\nHANDS_UNITS = 384\nPOSE_UNITS = 384\n# final embedding and transformer embedding size\nUNITS = 384\n\n# Transformer\nNUM_BLOCKS = 2\nMLP_RATIO = 2\n\n# Dropout\nEMBEDDING_DROPOUT = 0.00\nMLP_DROPOUT_RATIO = 0.30\nCLASSIFIER_DROPOUT_RATIO = 0.10\n\n# Initiailizers\nINIT_HE_UNIFORM = tf.keras.initializers.he_uniform\nINIT_GLOROT_UNIFORM = tf.keras.initializers.glorot_uniform\nINIT_ZEROS = tf.keras.initializers.constant(0.0)\n# Activations\nGELU = tf.keras.activations.gelu","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:18.777123Z","iopub.execute_input":"2023-05-06T16:07:18.777543Z","iopub.status.idle":"2023-05-06T16:07:18.785937Z","shell.execute_reply.started":"2023-05-06T16:07:18.777502Z","shell.execute_reply":"2023-05-06T16:07:18.784587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# based on: https://stackoverflow.com/questions/67342988/verifying-the-implementation-of-multihead-attention-in-transformer\n# replaced softmax with softmax layer to support masked softmax\ndef scaled_dot_product(q,k,v, softmax, attention_mask):\n    #calculates Q . K(transpose)\n    qkt = tf.matmul(q,k,transpose_b=True)\n    #caculates scaling factor\n    dk = tf.math.sqrt(tf.cast(q.shape[-1],dtype=tf.float32))\n    scaled_qkt = qkt/dk\n    softmax = softmax(scaled_qkt, mask=attention_mask)\n    \n    z = tf.matmul(softmax,v)\n    #shape: (m,Tx,depth), same shape as q,k,v\n    return z\n\nclass MultiHeadAttention(tf.keras.layers.Layer):\n    def __init__(self,d_model,num_of_heads):\n        super(MultiHeadAttention,self).__init__()\n        self.d_model = d_model\n        self.num_of_heads = num_of_heads\n        self.depth = d_model//num_of_heads\n        self.wq = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wk = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wv = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wo = tf.keras.layers.Dense(d_model)\n        self.softmax = tf.keras.layers.Softmax()\n        \n    def call(self,x, attention_mask):\n        \n        multi_attn = []\n        for i in range(self.num_of_heads):\n            Q = self.wq[i](x)\n            K = self.wk[i](x)\n            V = self.wv[i](x)\n            multi_attn.append(scaled_dot_product(Q,K,V, self.softmax, attention_mask))\n            \n        multi_head = tf.concat(multi_attn,axis=-1)\n        multi_head_attention = self.wo(multi_head)\n        return multi_head_attention","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:18.787966Z","iopub.execute_input":"2023-05-06T16:07:18.788393Z","iopub.status.idle":"2023-05-06T16:07:18.803191Z","shell.execute_reply.started":"2023-05-06T16:07:18.788305Z","shell.execute_reply":"2023-05-06T16:07:18.801804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Transformer(tf.keras.Model):\n    def __init__(self, num_blocks):\n        super(Transformer, self).__init__(name='transformer')\n        self.num_blocks = num_blocks\n    \n    def build(self, input_shape):\n        self.ln_1s = []\n        self.mhas = []\n        self.ln_2s = []\n        self.mlps = []\n        # Make Transformer Blocks\n        for i in range(self.num_blocks):\n            # First Layer Normalisation\n            self.ln_1s.append(tf.keras.layers.LayerNormalization(epsilon=LAYER_NORM_EPS))\n            # Multi Head Attention\n            self.mhas.append(MultiHeadAttention(UNITS, 12))\n            # Second Layer Normalisation\n            self.ln_2s.append(tf.keras.layers.LayerNormalization(epsilon=LAYER_NORM_EPS))\n            # Multi Layer Perception\n            self.mlps.append(tf.keras.Sequential([\n                tf.keras.layers.Dense(UNITS * MLP_RATIO, activation=GELU, kernel_initializer=INIT_GLOROT_UNIFORM),\n                tf.keras.layers.Dropout(MLP_DROPOUT_RATIO),\n                tf.keras.layers.Dense(UNITS, kernel_initializer=INIT_HE_UNIFORM),\n            ]))\n        \n    def call(self, x, attention_mask):\n        # Iterate input over transformer blocks\n        for ln_1, mha, ln_2, mlp in zip(self.ln_1s, self.mhas, self.ln_2s, self.mlps):\n            x1 = ln_1(x)\n            attention_output = mha(x1, attention_mask)\n            x2 = x1 + attention_output\n            x3 = ln_2(x2)\n            x3 = mlp(x3)\n            x = x3 + x2\n    \n        return x","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:18.806412Z","iopub.execute_input":"2023-05-06T16:07:18.806912Z","iopub.status.idle":"2023-05-06T16:07:18.820590Z","shell.execute_reply.started":"2023-05-06T16:07:18.806864Z","shell.execute_reply":"2023-05-06T16:07:18.819172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LandmarkEmbedding(tf.keras.Model):\n    def __init__(self, units, name):\n        super(LandmarkEmbedding, self).__init__(name=f'{name}_embedding')\n        self.units = units\n        \n    def build(self, input_shape):\n        # Embedding for missing landmark in frame, initizlied with zeros\n        self.empty_embedding = self.add_weight(\n            name=f'{self.name}_empty_embedding',\n            shape=[self.units],\n            initializer=INIT_ZEROS,\n        )\n        # Embedding\n        self.dense = tf.keras.Sequential([\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM, activation=GELU),\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name=f'{self.name}_dense')\n\n    def call(self, x):\n        return tf.where(\n                # Checks whether landmark is missing in frame\n                tf.reduce_sum(x, axis=2, keepdims=True) == 0,\n                # If so, the empty embedding is used\n                self.empty_embedding,\n                # Otherwise the landmark data is embedded\n                self.dense(x),\n            )","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:18.831227Z","iopub.execute_input":"2023-05-06T16:07:18.832006Z","iopub.status.idle":"2023-05-06T16:07:18.843020Z","shell.execute_reply.started":"2023-05-06T16:07:18.831959Z","shell.execute_reply":"2023-05-06T16:07:18.841578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Embedding","metadata":{}},{"cell_type":"code","source":"class CustomEmbedding(tf.keras.Model):\n    def __init__(self):\n        super(CustomEmbedding, self).__init__()\n        \n    def get_diffs(self, l):\n        S = l.shape[2]\n        other = tf.expand_dims(l, 3)\n        other = tf.repeat(other, S, axis=3)\n        other = tf.transpose(other, [0,1,3,2])\n        diffs = tf.expand_dims(l, 3) - other\n        diffs = tf.reshape(diffs, [-1, cfg.INPUT_SIZE, S*S])\n        return diffs\n\n    def build(self, input_shape):\n        # Positional Embedding, initialized with zeros\n        self.positional_embedding = tf.keras.layers.Embedding(cfg.INPUT_SIZE+1, UNITS, embeddings_initializer=INIT_ZEROS)\n        # Embedding layer for Landmarks\n        self.lips_embedding = LandmarkEmbedding(LIPS_UNITS, 'lips')\n        self.left_hand_embedding = LandmarkEmbedding(HANDS_UNITS, 'left_hand')\n        self.right_hand_embedding = LandmarkEmbedding(HANDS_UNITS, 'right_hand')\n        self.pose_embedding = LandmarkEmbedding(POSE_UNITS, 'pose')\n        # Landmark Weights\n        self.landmark_weights = tf.Variable(tf.zeros([4], dtype=tf.float32), name='landmark_weights')\n        # Fully Connected Layers for combined landmarks\n        self.fc = tf.keras.Sequential([\n            tf.keras.layers.Dense(UNITS, name='fully_connected_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM, activation=GELU),\n            tf.keras.layers.Dense(UNITS, name='fully_connected_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name='fc')\n\n\n    def call(self, lips0, left_hand0, right_hand0, pose0, non_empty_frame_idxs, training=False):\n        # Lips\n        lips_embedding = self.lips_embedding(lips0)\n        # Left Hand\n        left_hand_embedding = self.left_hand_embedding(left_hand0)\n        # Right Hand\n        right_hand_embedding = self.right_hand_embedding(right_hand0)\n        # Pose\n        pose_embedding = self.pose_embedding(pose0)\n        # Merge Embeddings of all landmarks with mean pooling\n        x = tf.stack((lips_embedding, left_hand_embedding, right_hand_embedding, pose_embedding), axis=3)\n        # Merge Landmarks with trainable attention weights\n        x = x * tf.nn.softmax(self.landmark_weights)\n        x = tf.reduce_sum(x, axis=3)\n        # Fully Connected Layers\n        x = self.fc(x)\n        # Add Positional Embedding\n        normalised_non_empty_frame_idxs = tf.where(\n            tf.math.equal(non_empty_frame_idxs, -1.0),\n            cfg.INPUT_SIZE,\n            tf.cast(\n                non_empty_frame_idxs / tf.reduce_max(non_empty_frame_idxs, axis=1, keepdims=True) * cfg.INPUT_SIZE,\n                tf.int32,\n            ),\n        )\n        x = x + self.positional_embedding(normalised_non_empty_frame_idxs)\n        \n        return x","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:18.844767Z","iopub.execute_input":"2023-05-06T16:07:18.845145Z","iopub.status.idle":"2023-05-06T16:07:18.863638Z","shell.execute_reply.started":"2023-05-06T16:07:18.845107Z","shell.execute_reply":"2023-05-06T16:07:18.862419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_lr_metric(optimizer):\n    def lr(y_true, y_pred):\n        return optimizer.lr\n    return lr","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:18.865040Z","iopub.execute_input":"2023-05-06T16:07:18.866094Z","iopub.status.idle":"2023-05-06T16:07:18.875896Z","shell.execute_reply.started":"2023-05-06T16:07:18.866056Z","shell.execute_reply":"2023-05-06T16:07:18.874667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    # Inputs\n    frames = tf.keras.layers.Input([cfg.INPUT_SIZE, N_COLS, cfg.N_DIMS], dtype=tf.float32, name='frames')\n    non_empty_frame_idxs = tf.keras.layers.Input([cfg.INPUT_SIZE], dtype=tf.float32, name='non_empty_frame_idxs')\n    # Padding Mask\n    mask = tf.cast(tf.math.not_equal(non_empty_frame_idxs, -1), tf.float32)\n    mask = tf.expand_dims(mask, axis=2)\n    \n    x = frames\n    x = tf.slice(x, [0,0,0,0], [-1,cfg.INPUT_SIZE, N_COLS, 2])\n    # LIPS\n    lips = tf.slice(x, [0,0,LIPS_START,0], [-1,cfg.INPUT_SIZE, 40, 2])\n    lips = tf.where(\n            tf.math.equal(lips, 0.0),\n            0.0,\n            (lips - LIPS_MEAN) / LIPS_STD,\n        )\n    lips = tf.reshape(lips, [-1, cfg.INPUT_SIZE, 40*2])\n    # LEFT HAND\n    left_hand = tf.slice(x, [0,0,40,0], [-1,cfg.INPUT_SIZE, 21, 2])\n    left_hand = tf.where(\n            tf.math.equal(left_hand, 0.0),\n            0.0,\n            (left_hand - LEFT_HANDS_MEAN) / LEFT_HANDS_STD,\n        )\n    left_hand = tf.reshape(left_hand, [-1, cfg.INPUT_SIZE, 21*2])\n    # RIGHT HAND\n    right_hand = tf.slice(x, [0,0,61,0], [-1,cfg.INPUT_SIZE, 21, 2])\n    right_hand = tf.where(\n            tf.math.equal(right_hand, 0.0),\n            0.0,\n            (right_hand - RIGHT_HANDS_MEAN) / RIGHT_HANDS_STD,\n        )\n    right_hand = tf.reshape(right_hand, [-1, cfg.INPUT_SIZE, 21*2])\n    # POSE\n    pose = tf.slice(x, [0,0,82,0], [-1,cfg.INPUT_SIZE, 10, 2])\n    pose = tf.where(\n            tf.math.equal(pose, 0.0),\n            0.0,\n            (pose - POSE_MEAN) / POSE_STD,\n        )\n    pose = tf.reshape(pose, [-1, cfg.INPUT_SIZE, 10*2])\n    x = lips, left_hand, right_hand, pose\n    x = CustomEmbedding()(lips, left_hand, right_hand, pose, non_empty_frame_idxs)\n    # Encoder Transformer Blocks\n    x = Transformer(NUM_BLOCKS)(x, mask)\n    # Pooling\n    x = tf.reduce_sum(x * mask, axis=1) / tf.reduce_sum(mask, axis=1)\n    # Classification Layer\n    x = tf.keras.layers.Dense(cfg.NUM_CLASSES, activation=tf.keras.activations.softmax, kernel_initializer=INIT_GLOROT_UNIFORM)(x)\n    outputs = x\n    \n    # Create Tensorflow Model\n    model = tf.keras.models.Model(inputs=[frames, non_empty_frame_idxs], outputs=outputs)\n    \n    # Simple Categorical Crossentropy Loss\n    loss = tf.keras.losses.SparseCategoricalCrossentropy()\n    \n    # Adam Optimizer with weight decay\n    optimizer = tfa.optimizers.AdamW(learning_rate=1e-3, weight_decay=1e-5, clipnorm=1.0)\n    \n    lr_metric = get_lr_metric(optimizer)\n    metrics = [\"acc\",lr_metric]\n    model.compile(loss=loss, optimizer=optimizer, metrics=metrics)\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:18.877189Z","iopub.execute_input":"2023-05-06T16:07:18.877560Z","iopub.status.idle":"2023-05-06T16:07:18.902567Z","shell.execute_reply.started":"2023-05-06T16:07:18.877526Z","shell.execute_reply":"2023-05-06T16:07:18.901302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nmodel_one = get_model()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:18.904103Z","iopub.execute_input":"2023-05-06T16:07:18.904787Z","iopub.status.idle":"2023-05-06T16:07:22.343911Z","shell.execute_reply.started":"2023-05-06T16:07:18.904743Z","shell.execute_reply":"2023-05-06T16:07:22.342723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_one.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:22.345512Z","iopub.execute_input":"2023-05-06T16:07:22.345985Z","iopub.status.idle":"2023-05-06T16:07:22.467955Z","shell.execute_reply.started":"2023-05-06T16:07:22.345939Z","shell.execute_reply":"2023-05-06T16:07:22.466538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Learning rate scheduler","metadata":{}},{"cell_type":"code","source":"def lrfn(current_step, num_warmup_steps, lr_max, num_cycles=0.50, num_training_steps=cfg.N_EPOCHS):\n    \n    if current_step < num_warmup_steps:\n        if WARMUP_METHOD == 'log':\n            return lr_max * 0.10 ** (num_warmup_steps - current_step)\n        else:\n            return lr_max * 2 ** -(num_warmup_steps - current_step)\n    else:\n        progress = float(current_step - num_warmup_steps) / float(max(1, num_training_steps - num_warmup_steps))\n\n        return max(0.0, 0.5 * (1.0 + math.cos(math.pi * float(num_cycles) * 2.0 * progress))) * lr_max","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:22.469842Z","iopub.execute_input":"2023-05-06T16:07:22.470350Z","iopub.status.idle":"2023-05-06T16:07:22.478068Z","shell.execute_reply.started":"2023-05-06T16:07:22.470277Z","shell.execute_reply":"2023-05-06T16:07:22.476604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_lr_schedule(lr_schedule, epochs):\n    fig = plt.figure(figsize=(20, 10))\n    plt.plot([None] + lr_schedule + [None])\n    # X Labels\n    x = np.arange(1, epochs + 1)\n    x_axis_labels = [i if epochs <= 40 or i % 5 == 0 or i == 1 else None for i in range(1, epochs + 1)]\n    plt.xlim([1, epochs])\n    plt.xticks(x, x_axis_labels) # set tick step to 1 and let x axis start at 1\n    \n    # Increase y-limit for better readability\n    plt.ylim([0, max(lr_schedule) * 1.1])\n    \n    # Title\n    schedule_info = f'start: {lr_schedule[0]:.1E}, max: {max(lr_schedule):.1E}, final: {lr_schedule[-1]:.1E}'\n    plt.title(f'Step Learning Rate Schedule, {schedule_info}', size=18, pad=12)\n    \n    # Plot Learning Rates\n    for x, val in enumerate(lr_schedule):\n        if epochs <= 40 or x % 5 == 0 or x is epochs - 1:\n            if x < len(lr_schedule) - 1:\n                if lr_schedule[x - 1] < val:\n                    ha = 'right'\n                else:\n                    ha = 'left'\n            elif x == 0:\n                ha = 'right'\n            else:\n                ha = 'left'\n            plt.plot(x + 1, val, 'o', color='black');\n            offset_y = (max(lr_schedule) - min(lr_schedule)) * 0.02\n            plt.annotate(f'{val:.1E}', xy=(x + 1, val + offset_y), size=12, ha=ha)\n    \n    plt.xlabel('Epoch', size=16, labelpad=5)\n    plt.ylabel('Learning Rate', size=16, labelpad=5)\n    plt.grid()\n    plt.show()\n\n# Learning rate for encoder\nLR_SCHEDULE = [lrfn(step, num_warmup_steps=cfg.N_WARMUP_EPOCHS, lr_max=cfg.LR_MAX, num_cycles=0.50) for step in range(cfg.N_EPOCHS)]\n# Plot Learning Rate Schedule\nplot_lr_schedule(LR_SCHEDULE, epochs=cfg.N_EPOCHS)\n# Learning Rate Callback\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lambda step: LR_SCHEDULE[step], verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:22.480135Z","iopub.execute_input":"2023-05-06T16:07:22.480662Z","iopub.status.idle":"2023-05-06T16:07:23.689871Z","shell.execute_reply.started":"2023-05-06T16:07:22.480609Z","shell.execute_reply":"2023-05-06T16:07:23.688833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Weight decay callback","metadata":{}},{"cell_type":"code","source":"# Custom callback to update weight decay with learning rate\nclass WeightDecayCallback(tf.keras.callbacks.Callback):\n    def __init__(self, wd_ratio=cfg.WD_RATIO):\n        self.step_counter = 0\n        self.wd_ratio = wd_ratio\n    \n    def on_epoch_begin(self, epoch, logs=None):\n        model_one.optimizer.weight_decay = model_one.optimizer.learning_rate * self.wd_ratio\n        print(f'learning rate: {model_one.optimizer.learning_rate.numpy():.2e}, weight decay: {model_one.optimizer.weight_decay.numpy():.2e}')","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:23.691432Z","iopub.execute_input":"2023-05-06T16:07:23.692565Z","iopub.status.idle":"2023-05-06T16:07:23.699944Z","shell.execute_reply.started":"2023-05-06T16:07:23.692520Z","shell.execute_reply":"2023-05-06T16:07:23.698554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%timeit -n 100\nif cfg.TRAIN_MODEL:\n    # Verify model_one prediction is <<<100ms\n    model_one.predict_on_batch({ 'frames': X[:1], 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS[:1] })","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:23.702062Z","iopub.execute_input":"2023-05-06T16:07:23.702510Z","iopub.status.idle":"2023-05-06T16:07:23.717560Z","shell.execute_reply.started":"2023-05-06T16:07:23.702468Z","shell.execute_reply":"2023-05-06T16:07:23.716415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Training","metadata":{}},{"cell_type":"code","source":"X_train = X[train_idxs]\nX_val = X[val_idxs]\nNON_EMPTY_FRAME_IDXS_TRAIN = NON_EMPTY_FRAME_IDXS[train_idxs]\nNON_EMPTY_FRAME_IDXS_VAL = NON_EMPTY_FRAME_IDXS[val_idxs]\ny_train = y[train_idxs]\ny_val = y[val_idxs]","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:23.719170Z","iopub.execute_input":"2023-05-06T16:07:23.720500Z","iopub.status.idle":"2023-05-06T16:07:25.510553Z","shell.execute_reply.started":"2023-05-06T16:07:23.720446Z","shell.execute_reply":"2023-05-06T16:07:25.509243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# delete variables as we go to free up RAM\ndel X_train; del y_train","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:25.515427Z","iopub.execute_input":"2023-05-06T16:07:25.516626Z","iopub.status.idle":"2023-05-06T16:07:25.529416Z","shell.execute_reply.started":"2023-05-06T16:07:25.516567Z","shell.execute_reply":"2023-05-06T16:07:25.527918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"if cfg.TRAIN_MODEL:\n    tf.keras.backend.clear_session()\n    callbacks=[\n            lr_callback,\n            WeightDecayCallback(),\n           # wandb.keras.WandbCallback()\n    ]\n    model_one.fit(\n        x=get_train_batch_all_signs(X, y, NON_EMPTY_FRAME_IDXS),\n        steps_per_epoch=len(X) // (cfg.NUM_CLASSES * cfg.BATCH_ALL_SIGNS_N),\n        epochs=cfg.N_EPOCHS,\n        batch_size=cfg.BATCH_SIZE,\n        callbacks=callbacks,\n        verbose = 2,) ","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:25.531327Z","iopub.execute_input":"2023-05-06T16:07:25.532027Z","iopub.status.idle":"2023-05-06T16:07:25.540803Z","shell.execute_reply.started":"2023-05-06T16:07:25.531974Z","shell.execute_reply":"2023-05-06T16:07:25.539820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save model artifacts\nif cfg.TRAIN_MODEL:# serialize weights to HDF5\n    model_one.save(\"model_v1.1.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:25.542163Z","iopub.execute_input":"2023-05-06T16:07:25.543167Z","iopub.status.idle":"2023-05-06T16:07:25.550150Z","shell.execute_reply.started":"2023-05-06T16:07:25.543127Z","shell.execute_reply":"2023-05-06T16:07:25.549010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Landmark Weights\n# weights = scipy.special.softmax(model_one.get_layer('custom_embedding').weights[15])\n# landmarks = ['lips_embedding', 'left_hand_embedding', 'right_hand_embedding', 'pose_embedding']\n\n# # Learned attention weights, initialized at uniform 25%\n# for w, lm in zip(weights, landmarks):\n#     print(f'{lm} weight: {(w*100):.1f}%')","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:25.551989Z","iopub.execute_input":"2023-05-06T16:07:25.552347Z","iopub.status.idle":"2023-05-06T16:07:25.561925Z","shell.execute_reply.started":"2023-05-06T16:07:25.552314Z","shell.execute_reply":"2023-05-06T16:07:25.560645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feed-Forward Neural Networks","metadata":{}},{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"\nNUM_FRAMES = 15\nSEGMENTS = 3\n\nLEFT_HAND_OFFSET = 468\nPOSE_OFFSET = LEFT_HAND_OFFSET+21\nRIGHT_HAND_OFFSET = POSE_OFFSET+33\n\n## average over the entire face, and the entire 'pose'\naveraging_sets = [[0, 468], [POSE_OFFSET, 33]]\n\nlip_landmarks = [61, 185, 40, 39, 37,  0, 267, 269, 270, 409,\n                 291,146, 91,181, 84, 17, 314, 405, 321, 375, \n                 78, 191, 80, 81, 82, 13, 312, 311, 310, 415, \n                 95, 88, 178, 87, 14,317, 402, 318, 324, 308]\nleft_hand_landmarks = list(range(LEFT_HAND_OFFSET, LEFT_HAND_OFFSET+21))\nright_hand_landmarks = list(range(RIGHT_HAND_OFFSET, RIGHT_HAND_OFFSET+21))\nprint(len(lip_landmarks))\npoint_landmarks = [item for sublist in [lip_landmarks, left_hand_landmarks, right_hand_landmarks] for item in sublist]\nLANDMARKS = len(point_landmarks) + len(averaging_sets)\nprint(LANDMARKS)\nif CFG.DROP_Z:\n    INPUT_SHAPE = (NUM_FRAMES,LANDMARKS*2)\nelse:\n    INPUT_SHAPE = (NUM_FRAMES,LANDMARKS*3)\n\nFLAT_INPUT_SHAPE = (INPUT_SHAPE[0] + 2 * (SEGMENTS + 1)) * INPUT_SHAPE[1]\nprint(FLAT_INPUT_SHAPE)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:25.563400Z","iopub.execute_input":"2023-05-06T16:07:25.563839Z","iopub.status.idle":"2023-05-06T16:07:25.579726Z","shell.execute_reply.started":"2023-05-06T16:07:25.563797Z","shell.execute_reply":"2023-05-06T16:07:25.578252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DROP_Z = True\n\nNUM_FRAMES = 15\nSEGMENTS = 3\n\nLEFT_HAND_OFFSET = 468\nPOSE_OFFSET = LEFT_HAND_OFFSET+21\nRIGHT_HAND_OFFSET = POSE_OFFSET+33\n\n## average over the entire face, and the entire 'pose'\naveraging_sets = [[0, 468], [POSE_OFFSET, 33]]\n\nlip_landmarks = [61, 185, 40, 39, 37,  0, 267, 269, 270, 409,\n                 291,146, 91,181, 84, 17, 314, 405, 321, 375, \n                 78, 191, 80, 81, 82, 13, 312, 311, 310, 415, \n                 95, 88, 178, 87, 14,317, 402, 318, 324, 308]\nleft_hand_landmarks = list(range(LEFT_HAND_OFFSET, LEFT_HAND_OFFSET+21))\nright_hand_landmarks = list(range(RIGHT_HAND_OFFSET, RIGHT_HAND_OFFSET+21))\n\npoint_landmarks = [item for sublist in [lip_landmarks, left_hand_landmarks, right_hand_landmarks] for item in sublist]\n\nLANDMARKS = len(point_landmarks) + len(averaging_sets)\nprint(LANDMARKS)\nif DROP_Z:\n    INPUT_SHAPE1 = (NUM_FRAMES,LANDMARKS*2)\nelse:\n    INPUT_SHAPE1 = (NUM_FRAMES,LANDMARKS*3)\n\nFLAT_INPUT_SHAPE = (INPUT_SHAPE1[0] + 2 * (SEGMENTS + 1)) * INPUT_SHAPE1[1]\nprint(FLAT_INPUT_SHAPE)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:25.581245Z","iopub.execute_input":"2023-05-06T16:07:25.582168Z","iopub.status.idle":"2023-05-06T16:07:25.597034Z","shell.execute_reply.started":"2023-05-06T16:07:25.582129Z","shell.execute_reply":"2023-05-06T16:07:25.595823Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utilites","metadata":{}},{"cell_type":"code","source":"def tf_nan_mean(x, axis=0):\n    return tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), x), axis=axis) / tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), tf.ones_like(x)), axis=axis)\n\ndef tf_nan_std(x, axis=0):\n    d = x - tf_nan_mean(x, axis=axis)\n    return tf.math.sqrt(tf_nan_mean(d * d, axis=axis))\n\ndef flatten_means_and_stds(x, axis=0):\n    # Get means and stds\n    x_mean = tf_nan_mean(x, axis=0)\n    x_std  = tf_nan_std(x,  axis=0)\n\n    x_out = tf.concat([x_mean, x_std], axis=0)\n    x_out = tf.reshape(x_out, (1, INPUT_SHAPE[1]*2))\n    x_out = tf.where(tf.math.is_finite(x_out), x_out, tf.zeros_like(x_out))\n    return x_out","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:25.598811Z","iopub.execute_input":"2023-05-06T16:07:25.599233Z","iopub.status.idle":"2023-05-06T16:07:25.611371Z","shell.execute_reply.started":"2023-05-06T16:07:25.599194Z","shell.execute_reply":"2023-05-06T16:07:25.609961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def flatten_means_and_stds1(x, axis=0):\n    # Get means and stds\n    x_mean = tf_nan_mean(x, axis=0)\n    x_std  = tf_nan_std(x,  axis=0)\n\n    x_out = tf.concat([x_mean, x_std], axis=0)\n    x_out = tf.reshape(x_out, (1, INPUT_SHAPE1[1]*2))\n    x_out = tf.where(tf.math.is_finite(x_out), x_out, tf.zeros_like(x_out))\n    return x_out\n\nclass FeatureGen_2(tf.keras.layers.Layer):\n    def __init__(self):\n        super(FeatureGen_2, self).__init__()\n    \n    def call(self, x_in):\n        if DROP_Z:\n            x_in = x_in[:, :, 0:2]\n        x_list = [tf.expand_dims(tf_nan_mean(x_in[:, av_set[0]:av_set[0]+av_set[1], :], axis=1), axis=1) for av_set in averaging_sets]\n        x_list.append(tf.gather(x_in, point_landmarks, axis=1))\n        x = tf.concat(x_list, 1)\n\n        x_padded = x\n        for i in range(SEGMENTS):\n            p0 = tf.where( ((tf.shape(x_padded)[0] % SEGMENTS) > 0) & ((i % 2) != 0) , 1, 0)\n            p1 = tf.where( ((tf.shape(x_padded)[0] % SEGMENTS) > 0) & ((i % 2) == 0) , 1, 0)\n            paddings = [[p0, p1], [0, 0], [0, 0]]\n            x_padded = tf.pad(x_padded, paddings, mode=\"SYMMETRIC\")\n        x_list = tf.split(x_padded, SEGMENTS)\n        x_list = [flatten_means_and_stds1(_x, axis=0) for _x in x_list]\n\n        x_list.append(flatten_means_and_stds1(x, axis=0))\n        \n        ## Resize only dimension 0. Resize can't handle nan, so replace nan with that dimension's avg value to reduce impact.\n        x = tf.image.resize(tf.where(tf.math.is_finite(x), x, tf_nan_mean(x, axis=0)), [NUM_FRAMES, LANDMARKS])\n        x = tf.reshape(x, (1, INPUT_SHAPE1[0]*INPUT_SHAPE1[1]))\n        x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n        x_list.append(x)\n        x = tf.concat(x_list, axis=1)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:25.613506Z","iopub.execute_input":"2023-05-06T16:07:25.613940Z","iopub.status.idle":"2023-05-06T16:07:25.630055Z","shell.execute_reply.started":"2023-05-06T16:07:25.613898Z","shell.execute_reply":"2023-05-06T16:07:25.628670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FeatureGen(tf.keras.layers.Layer):\n    def __init__(self):\n        super(FeatureGen, self).__init__()\n    \n    def call(self, x_in):\n        #x_in = load_relevant_data_subset(x_in)\n        if CFG.DROP_Z:\n            x_in = x_in[:, :, 0:2]\n        x_list = [tf.expand_dims(tf_nan_mean(x_in[:, av_set[0]:av_set[0]+av_set[1], :], axis=1), axis=1) \n                  for av_set in averaging_sets]\n        x_list.append(tf.gather(x_in, point_landmarks, axis=1))\n        x = tf.concat(x_list, 1)\n        #avg pose+face, 82 landmarks - (23,84,3)\n        x_padded = x\n        for i in range(SEGMENTS):\n            p0 = tf.where( ((tf.shape(x_padded)[0] % SEGMENTS) > 0) & ((i % 2) != 0) , 1, 0)\n            p1 = tf.where( ((tf.shape(x_padded)[0] % SEGMENTS) > 0) & ((i % 2) == 0) , 1, 0)\n            paddings = [[p0, p1], [0, 0], [0, 0]]\n            x_padded = tf.pad(x_padded, paddings, mode=\"SYMMETRIC\")\n        x_list = tf.split(x_padded, SEGMENTS)\n        #padding upto /3 and splitting - (24,84,3)\n\n        #flatten (8,84,3),(8,84,3),(8,84,3) with means and std - so, we would get 3 sets\n        x_list = [flatten_means_and_stds(_x, axis=0) for _x in x_list]\n\n        #flatten (23,84,3) with means and std\n        x_list.append(flatten_means_and_stds(x, axis=0))\n        print(x_list)\n        ## Resize only dimension 0. Resize can't handle nan, so replace nan with that dimension's avg value to reduce impact.\n        #resize (23,84,3) to (15,84,3)\n        x = tf.image.resize(tf.where(tf.math.is_finite(x), x, tf_nan_mean(x, axis=0)), [NUM_FRAMES, LANDMARKS])\n        #flatten it\n        x = tf.reshape(x, (1, INPUT_SHAPE[0]*INPUT_SHAPE[1]))\n        x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n        x_list.append(x)\n        x = tf.concat(x_list, axis=1)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:25.631567Z","iopub.execute_input":"2023-05-06T16:07:25.632210Z","iopub.status.idle":"2023-05-06T16:07:25.648130Z","shell.execute_reply.started":"2023-05-06T16:07:25.632171Z","shell.execute_reply":"2023-05-06T16:07:25.646865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#added in version 20\ndef normalize_data(x, axis=0):\n    # Get means and stds\n    x_mean = tf_nan_mean(x, axis=0)\n    x_std  = tf_nan_std(x,  axis=0)\n    return  (x - x_mean) / x_std","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:25.649365Z","iopub.execute_input":"2023-05-06T16:07:25.650412Z","iopub.status.idle":"2023-05-06T16:07:25.664125Z","shell.execute_reply.started":"2023-05-06T16:07:25.650370Z","shell.execute_reply":"2023-05-06T16:07:25.662847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create dataset","metadata":{}},{"cell_type":"code","source":"if CFG.make_featuregen:    \n    xs=[]\n    ys=[]\n    for i in tqdm(range(len(train))):\n        path = f\"{CFG.data_path}{train.iloc[i].path}\"\n        xs.append(FeatureGen()(load_relevant_data_subset(path)))\n        ys.append(train.iloc[i].label)\n    X = np.array(xs)\n    y = np.array(ys)\n    np.save(\"X_featuregen.npy\", X)\n    np.save(\"y_featuregen.npy\", y)\nelse:\n    if CFG.DROP_Z:\n        X = np.load(\"/kaggle/input/asl-features/feature-set-one/feature_data.npy\")\n        y = np.load(\"/kaggle/input/asl-features/feature-set-one/feature_labels.npy\")        \n    else:\n        X = np.load(\"/kaggle/input/islr-feature-gen-dataset/X_featuregen.npy\")\n        y = np.load(\"/kaggle/input/islr-feature-gen-dataset/y_featuregen.npy\")","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:25.666393Z","iopub.execute_input":"2023-05-06T16:07:25.667144Z","iopub.status.idle":"2023-05-06T16:07:39.802422Z","shell.execute_reply.started":"2023-05-06T16:07:25.667077Z","shell.execute_reply":"2023-05-06T16:07:39.801263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_norm_mean = tf_nan_mean(X, axis=0)\n# X_norm_std = tf_nan_std(X,  axis=0)\n\n# #data_leakage, split and then normalize\n# X=np.array(normalize_data(X))\n# X","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:39.804275Z","iopub.execute_input":"2023-05-06T16:07:39.805332Z","iopub.status.idle":"2023-05-06T16:07:39.811735Z","shell.execute_reply.started":"2023-05-06T16:07:39.805240Z","shell.execute_reply":"2023-05-06T16:07:39.810013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #don't use, could lead to data leakage\n# class FeatureGenInf(tf.keras.layers.Layer):\n#     def __init__(self, dist_mean = X_norm_mean, dist_std = X_norm_std ):\n#         super(FeatureGenInf, self).__init__()\n#         self.dist_mean = tf.constant(dist_mean,dtype=tf.float32)\n#         self.dist_std = tf.constant(dist_std, dtype=tf.float32)\n    \n#     def call(self, x_in):\n#         if CFG.DROP_Z:\n#             x_in = x_in[:, :, 0:2]\n#         x_list = [tf.expand_dims(tf_nan_mean(x_in[:, av_set[0]:av_set[0]+av_set[1], :], axis=1), axis=1) \n#                   for av_set in averaging_sets]\n#         x_list.append(tf.gather(x_in, point_landmarks, axis=1))\n#         x = tf.concat(x_list, 1)\n#         #avg pose+face, 82 landmarks - (23,84,3)\n#         x_padded = x\n#         for i in range(SEGMENTS):\n#             p0 = tf.where( ((tf.shape(x_padded)[0] % SEGMENTS) > 0) & ((i % 2) != 0) , 1, 0)\n#             p1 = tf.where( ((tf.shape(x_padded)[0] % SEGMENTS) > 0) & ((i % 2) == 0) , 1, 0)\n#             paddings = [[p0, p1], [0, 0], [0, 0]]\n#             x_padded = tf.pad(x_padded, paddings, mode=\"SYMMETRIC\")\n#         x_list = tf.split(x_padded, SEGMENTS)\n#         #padding upto /3 and splitting - (24,84,3)\n\n#         #flatten (8,84,3),(8,84,3),(8,84,3) with means and std - so, we would get n/3 sets\n#         x_list = [flatten_means_and_stds(_x, axis=0) for _x in x_list]\n\n#         #flatten (23,84,3) with means and std\n#         x_list.append(flatten_means_and_stds(x, axis=0))\n\n#         ## Resize only dimension 0. Resize can't handle nan, so replace nan with that dimension's avg value to reduce impact.\n#         #resize (23,84,3) to (15,84,3)\n#         x = tf.image.resize(tf.where(tf.math.is_finite(x), x, tf_nan_mean(x, axis=0)), [NUM_FRAMES, LANDMARKS])\n#         #flatten it\n#         x = tf.reshape(x, (1, INPUT_SHAPE[0]*INPUT_SHAPE[1]))\n#         x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n#         x_list.append(x)\n#         x = tf.concat(x_list, axis=1)\n#         x = (x - self.dist_mean) / self.dist_std\n#         return x","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:39.813505Z","iopub.execute_input":"2023-05-06T16:07:39.814876Z","iopub.status.idle":"2023-05-06T16:07:39.827152Z","shell.execute_reply.started":"2023-05-06T16:07:39.814741Z","shell.execute_reply":"2023-05-06T16:07:39.825749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def load_xy():\n#     all_x = np.load(\"/kaggle/input/isolated-sign-language-aggregation-preparation/feature_data.npy\").astype(np.float32)\n#     all_y = np.load(\"/kaggle/input/isolated-sign-language-aggregation-preparation/feature_labels.npy\").astype(np.uint8)\n\n#     # add nan back in not to screw up means/std\n#     all_x = np.where(all_x==0.0, np.nan, all_x)\n\n#     # Get mean and std ignoring nans\n#     all_mean = np.nanmean(all_x, keepdims=True, axis=0)\n#     all_std = np.nanstd(all_x, keepdims=True, axis=0)\n\n#     all_x = (all_x-all_mean)/all_std\n#     # Technically I don't need to do the were because np.nan \n#     # subtracting or dividing anything still results in np.nan\n#     #    - all_x = np.where(np.isnan(all_x), all_x, all_x-all_mean)\n#     #    - all_x = np.where(np.isnan(all_x), all_x, all_x/all_std)\n\n#     # Back to 0s\n#     all_x = np.nan_to_num(all_x)\n#     return all_x, all_y","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:39.828546Z","iopub.execute_input":"2023-05-06T16:07:39.828909Z","iopub.status.idle":"2023-05-06T16:07:39.843590Z","shell.execute_reply.started":"2023-05-06T16:07:39.828876Z","shell.execute_reply":"2023-05-06T16:07:39.841628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# all_x = np.load(\"/kaggle/input/isolated-sign-language-aggregation-preparation/feature_data.npy\").astype(np.float32)\n# all_y = np.load(\"/kaggle/input/isolated-sign-language-aggregation-preparation/feature_labels.npy\").astype(np.uint8)\n\n# # add nan back in not to screw up means/std\n# all_x = np.where(all_x==0.0, np.nan, all_x)\n\n# # Get mean and std ignoring nans\n# all_mean = np.nanmean(all_x, keepdims=True, axis=0)\n# all_std = np.nanstd(all_x, keepdims=True, axis=0)\n\n# all_x = (all_x-all_mean)/all_std\n# # Technically I don't need to do the were because np.nan \n# # subtracting or dividing anything still results in np.nan\n# #    - all_x = np.where(np.isnan(all_x), all_x, all_x-all_mean)\n# #    - all_x = np.where(np.isnan(all_x), all_x, all_x/all_std)\n\n# # Back to 0s\n# all_x = np.nan_to_num(all_x)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:39.845789Z","iopub.execute_input":"2023-05-06T16:07:39.846438Z","iopub.status.idle":"2023-05-06T16:07:39.859404Z","shell.execute_reply.started":"2023-05-06T16:07:39.846392Z","shell.execute_reply":"2023-05-06T16:07:39.858383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def dumb_tf_mean(x, axis=None):\n#     return tf.math.reduce_mean(x, axis=axis)\n\n# def dumb_tf_std(x, axis=None):\n#     x = tf.experimental.numpy.var(x, axis=axis, dtype=tf.float32, ddof=1)\n#     return tf.experimental.numpy.sqrt(x)\n\n# class PrepInputs(tf.keras.layers.Layer):\n#     def __init__(self, lh_idx_range=(468, 489), pose_idx_range=(489, 522), rh_idx_range=(522, 543), distribution_mean=all_mean, distribution_std=all_std):\n#         super(PrepInputs, self).__init__()\n#         self.lips = tf.constant([61, 185, 40, 39, 37, 0, 267, 269, 270, 409, 291, 146, 91, 181, 84, 17, 314, 405, 321, 375, 291, 78, 191, 80, 81, 82, 13, 312, 311, 310, 415, 308, 78, 95, 88, 178, 87, 14, 317, 402, 318, 324, 308])\n#         self.idx_ranges = [lh_idx_range, pose_idx_range, rh_idx_range]\n#         self.flat_feat_lens = [2*self.lips.shape[0],]+[2*(_range[1]-_range[0]) for _range in self.idx_ranges]\n#         self.distribution_mean = tf.constant(distribution_mean, dtype=tf.float32)\n#         self.distribution_std  = tf.constant(distribution_std, dtype=tf.float32)\n    \n#     def call(self, x_in):\n        \n#         # Split the single vector into 4\n#         xs = [tf.gather(x_in[..., :2], self.lips, axis=1),]+[x_in[:, _range[0]:_range[1], :2] for _range in self.idx_ranges]\n        \n#         # Reshape based on specific number of keypoints\n#         xs = [tf.reshape(_x, (-1, flat_feat_len)) for _x, flat_feat_len in zip(xs, self.flat_feat_lens)]\n        \n#         # Drop empty rows - Empty rows are present in \n#         #   --> face, lh, rh\n#         #   --> so we don't have to for face\n#         xs = [tf.boolean_mask(_x, tf.reduce_all(tf.logical_not(tf.math.is_nan(_x)), axis=1), axis=0) for _x in xs]\n        \n#         # Get means and stds\n#         x_means = [dumb_tf_mean(_x, axis=0) for _x in xs]\n#         x_stds  = [dumb_tf_std(_x,  axis=0) for _x in xs]\n        \n#         x_out = tf.concat([*x_means, *x_stds], axis=0)\n#         x_out = tf.expand_dims(tf.where(tf.math.is_nan(x_out), tf.zeros_like(x_out), x_out), axis=0)\n#         x_out = self.standardize_tensor(x_out)\n#         return x_out\n    \n#     def standardize_tensor(self, tensor):\n#         return tf.where(tensor!=0, (tensor-self.distribution_mean)/self.distribution_std, tf.zeros_like(tensor))\n    \n# p_demo = PrepInputs()(load_relevant_data_subset(CFG.data_path+train.path[0]))\n# print(p_demo.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:39.863816Z","iopub.execute_input":"2023-05-06T16:07:39.864528Z","iopub.status.idle":"2023-05-06T16:07:39.872787Z","shell.execute_reply.started":"2023-05-06T16:07:39.864470Z","shell.execute_reply":"2023-05-06T16:07:39.871525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Models","metadata":{}},{"cell_type":"code","source":"def get_model_2():\n    inputs = tf.keras.Input(shape=(472,), dtype=tf.float32)\n    vector = tf.keras.layers.Dense(1024, activation=\"gelu\")(inputs)\n    vector = tf.keras.layers.BatchNormalization()(vector) # Batch normalization layer\n    #vector = tf.keras.layers.Dropout(0.4)(vector) # Dropout layer\n    vector = tf.keras.layers.Reshape((1, 1024))(vector)\n    #vector - tf.keras.layers.LSTM(1024)(vector)\n    vector = tf.keras.layers.Dense(512, activation=\"gelu\")(vector)\n    vector = tf.keras.layers.BatchNormalization()(vector) # Batch normalization layer\n    #vector = tf.keras.layers.Dropout(0.4)(vector) # Dropout layer\n    vector = tf.keras.layers.Flatten()(vector)\n    output = tf.keras.layers.Dense(250, activation=\"softmax\")(vector)\n\n    model = tf.keras.Model(inputs=inputs, outputs=output)\n    model.compile(\n        optimizer = tf.keras.optimizers.Adam(learning_rate=1e-3),\n        loss=tf.keras.losses.SparseCategoricalCrossentropy(), \n        metrics=[\n            \"accuracy\", \n            tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5, name=\"top-5-accuracy\"),\n            tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10, name=\"top-10-accuracy\")\n        ]\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:39.874392Z","iopub.execute_input":"2023-05-06T16:07:39.875522Z","iopub.status.idle":"2023-05-06T16:07:39.891520Z","shell.execute_reply.started":"2023-05-06T16:07:39.875469Z","shell.execute_reply":"2023-05-06T16:07:39.890097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_1():\n    inputs = tf.keras.Input(shape=(1,5796), dtype=tf.float32)\n    vector = tf.keras.layers.Dense(1024, activation=\"gelu\")(inputs)\n    vector = tf.keras.layers.BatchNormalization()(vector) # Batch normalization layer\n    #vector = tf.keras.layers.Dropout(0.2)(vector) # Dropout layer\n    vector = tf.keras.layers.Dense(512, activation=\"gelu\")(vector)\n    vector = tf.keras.layers.BatchNormalization()(vector) # Batch normalization layer\n    #vector = tf.keras.layers.Dropout(0.2)(vector) # Dropout layer\n    vector = tf.keras.layers.Flatten()(vector)\n    output = tf.keras.layers.Dense(250, activation=\"softmax\")(vector)\n\n    model = tf.keras.Model(inputs=inputs, outputs=output)\n    model.compile(\n        optimizer = tf.keras.optimizers.Adam(learning_rate=1e-3),\n        loss=tf.keras.losses.SparseCategoricalCrossentropy(), \n        metrics=[\n            \"accuracy\", \n            tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5, name=\"top-5-accuracy\"),\n            tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10, name=\"top-10-accuracy\")\n        ]\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:39.893594Z","iopub.execute_input":"2023-05-06T16:07:39.894510Z","iopub.status.idle":"2023-05-06T16:07:39.906987Z","shell.execute_reply.started":"2023-05-06T16:07:39.894468Z","shell.execute_reply":"2023-05-06T16:07:39.905799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model_4():\n    inputs = tf.keras.Input(shape=(3864,), dtype=tf.float32)\n    vector = tf.keras.layers.Dense(2048)(inputs)\n    vector = tf.keras.layers.BatchNormalization()(vector)\n    vector = tf.keras.layers.Activation('gelu')(vector)\n    #vector = tf.keras.layers.Dropout(0.3)(vector)\n\n    vector = tf.keras.layers.Dense(1024)(vector)\n    vector = tf.keras.layers.BatchNormalization()(vector)\n    vector = tf.keras.layers.Activation('gelu')(vector)\n    #vector = tf.keras.layers.Dropout(0.1)(vector)\n\n    vector = tf.keras.layers.Dense(512)(vector)\n    vector = tf.keras.layers.BatchNormalization()(vector)\n    vector = tf.keras.layers.Activation('gelu')(vector)\n    #vector = tf.keras.layers.Dropout(0.1)(vector)\n    vector = tf.keras.layers.Flatten()(vector)\n    outputs = tf.keras.layers.Dense(250, activation=\"softmax\")(vector)\n\n    model = tf.keras.Model(inputs=inputs, outputs=outputs)\n    model.compile(\n        optimizer = tf.keras.optimizers.Adam(learning_rate=1e-3),\n        loss=tf.keras.losses.SparseCategoricalCrossentropy(), \n        metrics=[\n            \"accuracy\", \n            tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5, name=\"top-5-accuracy\"),\n            tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10, name=\"top-10-accuracy\")\n        ]\n    )\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:39.908992Z","iopub.execute_input":"2023-05-06T16:07:39.909892Z","iopub.status.idle":"2023-05-06T16:07:39.926068Z","shell.execute_reply.started":"2023-05-06T16:07:39.909832Z","shell.execute_reply":"2023-05-06T16:07:39.924536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # define the K-fold cross-validation\n# k = 4\n# gkf = GroupKFold(n_splits=k)\n# groups=train['participant_id'].values\n\n# X_fold = np.array(X).reshape(X.shape[0],X.shape[2])\n# # initialize empty lists to store the validation accuracy and best model checkpoints for each fold\n# val_acc_per_fold = []\n# best_model_per_fold = []\n\n# gc.collect()\n# # loop over the folds\n# if CFG.kfold_training:\n#     for fold, (train_idx, val_idx) in enumerate(gkf.split(all_x, all_y, groups)):\n#         print(f\"Fold {fold+1}...\")\n#         X_train, y_train = all_x[train_idx], all_y[train_idx]\n#         X_val, y_val = all_x[val_idx], all_y[val_idx]\n\n#         # define the callbacks for this fold\n#         callbacks = [\n#             tf.keras.callbacks.ReduceLROnPlateau(monitor='val_accuracy', factor=0.97,\n#                                                  patience=2, verbose=0),\n#             tf.keras.callbacks.ModelCheckpoint(f\"modelv0.8.1_fold_{fold}.h5\", save_best_only=True, \n#                                                restore_best_weights=True, monitor=\"val_accuracy\", \n#                                                mode=\"max\")\n#         ]\n\n#         # create and train the model\n#         model = get_model_2()\n#         history = model.fit(X_train, y_train, epochs=60, validation_data=(X_val, y_val), \n#                             batch_size=1024, callbacks=callbacks)\n        \n#         del X_train, X_val\n#         gc.collect()\n#         # evaluate the model on the validation data and save the best model checkpoint\n#         val_acc = np.max(history.history[\"val_accuracy\"])\n#         print(f\"Val_acc_fold{fold}:{val_acc}\")\n#         val_acc_per_fold.append(val_acc)\n#         best_model_per_fold.append(tf.keras.models.load_model(f\"modelv0.8.1_fold_{fold}.h5\"))\n# # else:\n# #     for fold in range(k):\n# #         best_model_per_fold.append(tf.keras.models.load_model(f\"/kaggle/input/islr-submission-models/modelv0.8_fold_{fold}.h5\"))\n\n# del all_x,all_y,X_fold\n# gc.collect()\n\n# def get_model_3():\n#     inputs = tf.keras.Input(shape=(472,), dtype=tf.float32)\n#     outputs = [model(inputs) for model in best_model_per_fold]\n#     output = tf.keras.layers.Average()(outputs)\n    \n#     model = tf.keras.Model(inputs=inputs, outputs=output)\n#     model.compile(\n#         optimizer = tf.keras.optimizers.Adam(learning_rate=0.00033),\n#         loss=tf.keras.losses.SparseCategoricalCrossentropy(), \n#         metrics=[\n#             \"accuracy\", \n#             tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5, name=\"top-5-accuracy\"),\n#             tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10, name=\"top-10-accuracy\")\n#         ]\n#     )\n\n#     return model\n\n# #get_model_3().save(\"model_v0.8.1.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:39.930731Z","iopub.execute_input":"2023-05-06T16:07:39.932490Z","iopub.status.idle":"2023-05-06T16:07:39.944471Z","shell.execute_reply.started":"2023-05-06T16:07:39.932430Z","shell.execute_reply":"2023-05-06T16:07:39.942757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"if CFG.is_training:\n    #X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=33, stratify=y)\n    #print(X_train.shape, y_train.shape, X_val.shape, y_val.shape)\n    #del X,y\n    gc.collect()\n    model = get_model_4()\n#     callbacks = [\n#         tf.keras.callbacks.ReduceLROnPlateau(monitor='val_accuracy', factor=0.97,\n#                                               patience=2, verbose=0),\n#         tf.keras.callbacks.ModelCheckpoint(\"model_linear57.h5\", save_best_only=True, \n#                                            restore_best_weights=True, monitor=\"val_accuracy\", \n#                                            mode=\"max\")\n#     ]\n    callbacks = [\n        tf.keras.callbacks.ReduceLROnPlateau(monitor='accuracy', factor=0.97,\n                                              patience=2, verbose=0),\n        tf.keras.callbacks.ModelCheckpoint(\"model_linear38.h5\", save_best_only=True, \n                                           restore_best_weights=True, monitor=\"accuracy\", \n                                           mode=\"max\")\n    ]\n#     model.fit(X_train, y_train, epochs=100, validation_data=(X_val, y_val), \n#               batch_size=256, callbacks=callbacks)\n    model.fit(X, y, epochs=120, batch_size=64, callbacks=callbacks)\n#     model.fit(X, y, epochs=100, validation_data=(X_val, y_val), \n#               batch_size=256, callbacks=callbacks)\n    \n    # Plot the validation accuracy over each epoch\n#     plt.plot(model.history.history[\"val_accuracy\"])\n#     plt.title(\"Validation Accuracy Over Epochs\")\n#     plt.xlabel(\"Epoch\")\n#     plt.ylabel(\"Validation Accuracy\")\n#     plt.show()\n    \nelse:\n    #model_0 = tf.keras.models.load_model(\"/kaggle/input/islr-submission-models/model_v0.2.h5\")\n    model_1 = tf.keras.models.load_model(\"/kaggle/input/islr-submission-models/model_linear57.h5\")\n    #model_2 = tf.keras.models.load_model(\"/kaggle/input/islr-submission-models/model_v0.4.2.h5\")\n    #model_3 = tf.keras.models.load_model(\"/kaggle/input/islr-submission-models/model_v0.8.1.h5\")\n    model_4 = tf.keras.models.load_model(\"/kaggle/input/islr-submission-models/model_linear38.h5\")\n    \n    model_5 = tf.keras.models.load_model(\"/kaggle/input/islr-submission-models/model_v1.1.h5\", \n                                         custom_objects = {'CustomEmbedding': CustomEmbedding, \n                                                           'Transformer': lambda: Transformer(num_blocks=2),\n                                                           'lr': get_lr_metric})\n\n#model_1.summary()\n# model_2.summary()\n#model_3.summary()\n# model_4.summary()\n# model_5.summary()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:39.946031Z","iopub.execute_input":"2023-05-06T16:07:39.947686Z","iopub.status.idle":"2023-05-06T16:07:47.539733Z","shell.execute_reply.started":"2023-05-06T16:07:39.947621Z","shell.execute_reply":"2023-05-06T16:07:47.538384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create an Inference Model\nThis inference Model wraps the previous trained DNN model and do following preprocessing:\n* Replace nan value with 0\n* calcuate mean frame of the input tensor with (None, 543, 3) shape and convert to (1, 543, 3) shape.","metadata":{}},{"cell_type":"code","source":"def get_inference_model(model_1, model_4, model_5):\n    inputs = tf.keras.Input((543, 3), dtype=tf.float32, name=\"inputs\")\n    #x0 = tf.gather(inputs, LANDMARK_IDX, axis=1)\n    #x0 = tf.where(tf.math.is_nan(x0), tf.zeros_like(x0), x0)\n    #x0 = tf.reduce_mean(x0, axis=0, keepdims=True)\n    x1 = FeatureGen()(inputs)\n    #x2 = PrepInputs()(inputs)\n    #x3 = PrepInputs()(inputs)\n    x4 = FeatureGen_2()(inputs)\n    x5, non_empty_frame_idxs  = PreprocessLayer()(inputs)\n    \n    x1 = tf.reshape(x1, (-1, 1, 5796))\n    #x2 = tf.reshape(x2, (-1,472))\n    #x3 = tf.reshape(x3, (-1,472))\n    #x4 = tf.reshape(x4, (-1, 1, 3864))\n    x5 = tf.expand_dims(x5, axis=0)\n    non_empty_frame_idxs = tf.expand_dims(non_empty_frame_idxs, axis=0)\n    \n    x1_out = model_1(x1)[0, :]\n    #x2 = model_2(x2)\n    #x3 = model_3(x3)\n    #x0 = model_0(x0)\n    x4_out = model_4(x4)[0, :]\n    x5_out = model_5({ 'frames': x5, 'non_empty_frame_idxs': non_empty_frame_idxs })\n    x5_out = tf.squeeze(x5_out, axis=0)\n    \n    output = (0.1 * x1_out) + (0.2* x4_out) + (0.7* x5_out)\n    #x = tf.keras.layers.Average()([x1,x2])\n    output = tf.keras.layers.Activation(activation=\"linear\", name=\"outputs\")(output)\n    inference_model = tf.keras.Model(inputs=inputs, outputs=output) \n    inference_model.compile(loss=tf.keras.losses.SparseCategoricalCrossentropy(), metrics=[\"accuracy\"])\n    return inference_model","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:47.542330Z","iopub.execute_input":"2023-05-06T16:07:47.542775Z","iopub.status.idle":"2023-05-06T16:07:47.554854Z","shell.execute_reply.started":"2023-05-06T16:07:47.542732Z","shell.execute_reply":"2023-05-06T16:07:47.553396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_model = get_inference_model(model_1, model_4, model_5)\n#inference_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:47.557080Z","iopub.execute_input":"2023-05-06T16:07:47.557505Z","iopub.status.idle":"2023-05-06T16:07:50.428581Z","shell.execute_reply.started":"2023-05-06T16:07:47.557467Z","shell.execute_reply":"2023-05-06T16:07:50.427373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_model.save(\"ensemble_model_5796linear+3864Linear+Transformer.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:13:37.332111Z","iopub.execute_input":"2023-05-06T16:13:37.332626Z","iopub.status.idle":"2023-05-06T16:13:37.679351Z","shell.execute_reply.started":"2023-05-06T16:13:37.332586Z","shell.execute_reply":"2023-05-06T16:13:37.677874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create submission file\nUnlike a csv file in our previous competitions, the submission file has to be a compressed tflite file with name submissiom.zip.","metadata":{}},{"cell_type":"code","source":"converter = tf.lite.TFLiteConverter.from_keras_model(inference_model)\nconverter.optimizations = [tf.lite.Optimize.DEFAULT] #great line\ntflite_model = converter.convert()\nmodel_path = \"ensemble_model_5796linear+3864Linear+Transformer.tflite\"\n# Save the model.\nwith open(model_path, 'wb') as f:\n    f.write(tflite_model)\nos.path.getsize(\"ensemble_model_5796linear+3864Linear+Transformer.tflite\")","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:07:50.430117Z","iopub.execute_input":"2023-05-06T16:07:50.430642Z","iopub.status.idle":"2023-05-06T16:09:34.627394Z","shell.execute_reply.started":"2023-05-06T16:07:50.430588Z","shell.execute_reply":"2023-05-06T16:09:34.626020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission.zip $model_path","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:09:34.629000Z","iopub.execute_input":"2023-05-06T16:09:34.629353Z","iopub.status.idle":"2023-05-06T16:09:37.194110Z","shell.execute_reply.started":"2023-05-06T16:09:34.629319Z","shell.execute_reply":"2023-05-06T16:09:37.192517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making inferences\nLet's make inferences using TFLite interpreter, our submission file will be used to make inference like following ways. To test inference speed, I will make inferences with 10000 samples.","metadata":{}},{"cell_type":"code","source":"!pip install tflite-runtime","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:14:18.936733Z","iopub.execute_input":"2023-05-06T16:14:18.937550Z","iopub.status.idle":"2023-05-06T16:14:32.131834Z","shell.execute_reply.started":"2023-05-06T16:14:18.937482Z","shell.execute_reply":"2023-05-06T16:14:32.130368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tflite_runtime.interpreter as tflite\ninterpreter = tflite.Interpreter(model_path)\nfound_signatures = list(interpreter.get_signature_list().keys())\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")\nfor i in tqdm(range(1000)):\n    frames = load_relevant_data_subset(f'/kaggle/input/asl-signs/{train.iloc[i].path}')\n    output = prediction_fn(inputs=frames)\n    sign = np.argmax(output[\"outputs\"])\n    if i % 100 == 0:\n        print(f\"Predicted label: {index_label[sign]}, Actual Label: {train.iloc[i].sign}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-06T16:16:53.257407Z","iopub.execute_input":"2023-05-06T16:16:53.257950Z","iopub.status.idle":"2023-05-06T16:17:59.088458Z","shell.execute_reply.started":"2023-05-06T16:16:53.257904Z","shell.execute_reply":"2023-05-06T16:17:59.086994Z"},"trusted":true},"execution_count":null,"outputs":[]}]}