{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\nimport seaborn as sn\n\nfrom tensorflow import keras\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split, GroupShuffleSplit, StratifiedGroupKFold\n\nimport glob\nimport sys\nimport os\nimport math\nimport gc\nimport sys\nimport sklearn\nimport scipy\nimport random\nfrom types import SimpleNamespace\n\nprint(f'Tensorflow V{tf.__version__}')\nprint(f'Keras V{tf.keras.__version__}')\nprint(f'Python V{sys.version}')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:03.228779Z","iopub.execute_input":"2023-05-01T13:31:03.229564Z","iopub.status.idle":"2023-05-01T13:31:12.747543Z","shell.execute_reply.started":"2023-05-01T13:31:03.229520Z","shell.execute_reply":"2023-05-01T13:31:12.746125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# matplotlib参数配置","metadata":{}},{"cell_type":"code","source":"# MatplotLib Global Settings\n# 重置所有参数为默认值\nplt.rcParams.update(mpl.rcParamsDefault)\n# 设置 x 轴和 y 轴的刻度标签字体大小为 16\nplt.rcParams['xtick.labelsize'] = 16\nplt.rcParams['ytick.labelsize'] = 16\n# 设置坐标轴标签和标题的字体大小为 18 和 24\nplt.rcParams['axes.labelsize'] = 18\nplt.rcParams['axes.titlesize'] = 24\n","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:12.753637Z","iopub.execute_input":"2023-05-01T13:31:12.756386Z","iopub.status.idle":"2023-05-01T13:31:12.768070Z","shell.execute_reply.started":"2023-05-01T13:31:12.756344Z","shell.execute_reply":"2023-05-01T13:31:12.766784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#cfg = SimpleNamespace()\n#iskaggle = os.environ.get('KAGGLE_KERNEL_RUN_TYPE', '')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:12.773169Z","iopub.execute_input":"2023-05-01T13:31:12.775808Z","iopub.status.idle":"2023-05-01T13:31:12.782783Z","shell.execute_reply.started":"2023-05-01T13:31:12.775760Z","shell.execute_reply":"2023-05-01T13:31:12.781434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from pathlib import Path\n# DATA_DIR         = Path('../data/') if not iskaggle else Path('/kaggle/input/asl-signs/')\n# TRAIN_CSV_PATH   = DATA_DIR/'train.csv'\n# LANDMARK_DIR     = DATA_DIR/'train_landmark_files'\n# LABEL_MAP_PATH   = DATA_DIR/'sign_to_prediction_index_map.json'","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:12.789549Z","iopub.execute_input":"2023-05-01T13:31:12.791831Z","iopub.status.idle":"2023-05-01T13:31:12.798165Z","shell.execute_reply.started":"2023-05-01T13:31:12.791785Z","shell.execute_reply":"2023-05-01T13:31:12.796942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 训练参数配置","metadata":{}},{"cell_type":"code","source":"# If True, processing data from scratch\n# If False, loads preprocessed data\n# True，则从头开始处理数据，False，则加载预处理的数据\n\nPREPROCESS_DATA = False\nTRAIN_MODEL = True\n# True: use 10% of participants as validation set\n# False: use all data for training -> gives better LB result更好的leaderboard排名\nUSE_VAL = False\n\nN_ROWS = 543#设置行数\nN_DIMS = 3#设置维度\nDIM_NAMES = ['x', 'y', 'z']#设置维度名称\nSEED = 42#设置随机数种子\nNUM_CLASSES = 250#共250种类别\nIS_INTERACTIVE = os.environ['KAGGLE_KERNEL_RUN_TYPE'] == 'Interactive'\n#判断是否为交互式环境\nVERBOSE = 1 if IS_INTERACTIVE else 2\n\nINPUT_SIZE = 64#输入大小\nBATCH_ALL_SIGNS_N = 4\nBATCH_SIZE = 512\nN_EPOCHS = 1\nLR_MAX = 0.001#设置学习率最大值\nN_WARMUP_EPOCHS = 0#设置热身轮数\nWD_RATIO = 0.044#设置权重衰减比例\nMASK_VAL = 4237#设置掩码值","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:12.803192Z","iopub.execute_input":"2023-05-01T13:31:12.804059Z","iopub.status.idle":"2023-05-01T13:31:12.815907Z","shell.execute_reply.started":"2023-05-01T13:31:12.804023Z","shell.execute_reply":"2023-05-01T13:31:12.814973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Prints Shape and Dtype For List Of Variables\ndef print_shape_dtype(l, names):\n    for e, n in zip(l, names):\n        print(f'{n} shape: {e.shape}, dtype: {e.dtype}')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:12.820760Z","iopub.execute_input":"2023-05-01T13:31:12.823679Z","iopub.status.idle":"2023-05-01T13:31:12.831213Z","shell.execute_reply.started":"2023-05-01T13:31:12.823644Z","shell.execute_reply":"2023-05-01T13:31:12.830254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"遍历变量列表和名称列表，然后打印每个变量的形状和数据类型","metadata":{}},{"cell_type":"markdown","source":"# 读取训练数据","metadata":{}},{"cell_type":"code","source":"# Read Training Data\n\nif IS_INTERACTIVE or not PREPROCESS_DATA:\n    train = pd.read_csv('/kaggle/input/asl-signs/train.csv').sample(int(5e3), random_state=SEED)\nelse:\n    train = pd.read_csv('/kaggle/input/asl-signs/train.csv')\n\nN_SAMPLES = len(train)\nprint(f'N_SAMPLES: {N_SAMPLES}')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:12.834660Z","iopub.execute_input":"2023-05-01T13:31:12.835382Z","iopub.status.idle":"2023-05-01T13:31:13.037936Z","shell.execute_reply.started":"2023-05-01T13:31:12.835347Z","shell.execute_reply":"2023-05-01T13:31:13.035854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 添加文件路径","metadata":{}},{"cell_type":"markdown","source":"这段代码定义了一个名为get_file_path()的函数，它接受一个参数path，并返回一个完整的文件路径。该函数用于将文件路径转换为Kaggle数据集中的文件路径。\n代码使用Pandas的apply()函数将get_file_path()函数应用于名为train['path']的Pandas Series对象中的每个元素。然后，代码将结果存储在名为train['file_path']的新列中。","metadata":{}},{"cell_type":"code","source":"# Get complete file path to file\ndef get_file_path(path):\n    return f'/kaggle/input/asl-signs/{path}'\n\ntrain['file_path'] = train['path'].apply(get_file_path)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:13.040481Z","iopub.execute_input":"2023-05-01T13:31:13.041118Z","iopub.status.idle":"2023-05-01T13:31:13.053456Z","shell.execute_reply.started":"2023-05-01T13:31:13.041072Z","shell.execute_reply":"2023-05-01T13:31:13.052242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 对符号进行数字编码","metadata":{}},{"cell_type":"code","source":"# Add ordinally Encoded Sign (assign number to each sign name)\n# 添加序数编码符号（为每个符号名称分配编号）\ntrain['sign_ord'] = train['sign'].astype('category').cat.codes\n\n# Dictionaries to translate sign <-> ordinal encoded sign\n# 用于符号<->序数编码之间相互转换的字典\nSIGN2ORD = train[['sign', 'sign_ord']].set_index('sign').squeeze().to_dict()\nORD2SIGN = train[['sign_ord', 'sign']].set_index('sign_ord').squeeze().to_dict()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:13.055353Z","iopub.execute_input":"2023-05-01T13:31:13.055714Z","iopub.status.idle":"2023-05-01T13:31:13.080496Z","shell.execute_reply.started":"2023-05-01T13:31:13.055677Z","shell.execute_reply":"2023-05-01T13:31:13.079564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(train.head(3))\ndisplay(train.info())","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:13.085038Z","iopub.execute_input":"2023-05-01T13:31:13.085308Z","iopub.status.idle":"2023-05-01T13:31:13.117008Z","shell.execute_reply.started":"2023-05-01T13:31:13.085283Z","shell.execute_reply":"2023-05-01T13:31:13.115939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 统计视频数据","metadata":{}},{"cell_type":"markdown","source":"这段代码是用于计算视频数据集中每个视频的不同帧数、缺失帧数和最大帧数的统计信息。\n其中，`N` 是根据条件设置的值，`N_UNIQUE_FRAMES`、`N_MISSING_FRAMES` 和 `MAX_FRAME` 分别是存储不同帧数、缺失帧数和最大帧数的数组。\n在循环中，代码会遍历数据集中的每个视频，读取视频数据并计算不同帧数、缺失帧数和最大帧数。\n最后，代码会显示这些统计信息，并绘制相应的直方图。","metadata":{}},{"cell_type":"code","source":"# N_PARTICIPANTS = train.participant_id.nunique()\n# sgkf = StratifiedGroupKFold(n_splits=7, shuffle=True, random_state=43)\n# train['fold'] = -1\n# for i, (train_idx, val_idx) in enumerate(sgkf.split(train.index, train.sign, train.participant_id)):\n#     train.loc[val_idx, 'fold'] = i\n# train.head(2)\n\n# # create indexes using fold `0` for now\n# train_idxs = train.query(\"fold!=0\").index.values\n# val_idxs = train.query(\"fold==0\").index.values\n# len(train_idxs), len(val_idxs)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:13.118583Z","iopub.execute_input":"2023-05-01T13:31:13.119171Z","iopub.status.idle":"2023-05-01T13:31:13.124466Z","shell.execute_reply.started":"2023-05-01T13:31:13.119133Z","shell.execute_reply":"2023-05-01T13:31:13.123319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N = int(1e3) if (IS_INTERACTIVE or not PREPROCESS_DATA) else int(10e3)  # 根据条件设置 N 的值\nN_UNIQUE_FRAMES = np.zeros(N, dtype=np.uint16)  # 初始化 N_UNIQUE_FRAMES 数组\nN_MISSING_FRAMES = np.zeros(N, dtype=np.uint16)  # 初始化 N_MISSING_FRAMES 数组\nMAX_FRAME = np.zeros(N, dtype=np.uint16)  # 初始化 MAX_FRAME 数组\n\nPERCENTILES = [0.01, 0.05, 0.25, 0.50, 0.75, 0.95, 0.99, 0.999]  # 定义 PERCENTILES 列表\n\nfor idx, file_path in enumerate(tqdm(train['file_path'].sample(N, random_state=SEED))):  # 遍历 train['file_path'] 中的文件路径\n    df = pd.read_parquet(file_path)  # 读取文件\n    N_UNIQUE_FRAMES[idx] = df['frame'].nunique()  # 计算每个文件中不同帧的数量\n    N_MISSING_FRAMES[idx] = (df['frame'].max() - df['frame'].min()) - df['frame'].nunique() + 1  # 计算每个文件中缺失帧的数量\n    MAX_FRAME[idx] = df['frame'].max()  # 计算每个文件中最大帧数","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:13.126065Z","iopub.execute_input":"2023-05-01T13:31:13.126489Z","iopub.status.idle":"2023-05-01T13:31:36.810167Z","shell.execute_reply.started":"2023-05-01T13:31:13.126429Z","shell.execute_reply":"2023-05-01T13:31:36.808882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of unique frames in each video\ndisplay(pd.Series(N_UNIQUE_FRAMES).describe(percentiles=PERCENTILES).to_frame('N_UNIQUE_FRAMES'))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Unique Frames', size=24)\npd.Series(N_UNIQUE_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nxlim = math.ceil(plt.xlim()[1])\nplt.xlim(0, xlim)\nplt.xticks(np.arange(0, xlim+25, 25))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:36.811796Z","iopub.execute_input":"2023-05-01T13:31:36.812885Z","iopub.status.idle":"2023-05-01T13:31:37.349999Z","shell.execute_reply.started":"2023-05-01T13:31:36.812843Z","shell.execute_reply":"2023-05-01T13:31:37.348878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of missing frames, consecutive frames with missing intermediate frame, i.e. 1,2,4,5 -> 3 is missing\n#丢失帧数，丢失中间帧的连续帧\ndisplay(pd.Series(N_MISSING_FRAMES).describe(percentiles=PERCENTILES).to_frame('N_MISSING_FRAMES'))\n\nplt.figure(figsize=(15,8))\nplt.title('Number of Missing Frames', size=24)\npd.Series(N_MISSING_FRAMES).plot(kind='hist', bins=128)\nplt.grid()\nplt.xlim(0, math.ceil(plt.xlim()[1]))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:37.351657Z","iopub.execute_input":"2023-05-01T13:31:37.352033Z","iopub.status.idle":"2023-05-01T13:31:37.797156Z","shell.execute_reply.started":"2023-05-01T13:31:37.351996Z","shell.execute_reply":"2023-05-01T13:31:37.796145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Maximum frame number\ndisplay(pd.Series(MAX_FRAME).describe(percentiles=PERCENTILES).to_frame('MAX_FRAME'))\n\nplt.figure(figsize=(15,8))\nplt.title('Maximum Frames Index', size=24)\npd.Series(MAX_FRAME).plot(kind='hist', bins=128)\nplt.grid()\nplt.xlim(0, math.ceil(plt.xlim()[1]))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:37.798857Z","iopub.execute_input":"2023-05-01T13:31:37.799628Z","iopub.status.idle":"2023-05-01T13:31:38.236865Z","shell.execute_reply.started":"2023-05-01T13:31:37.799580Z","shell.execute_reply":"2023-05-01T13:31:38.235901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Landmark Indices 关键点索引","metadata":{}},{"cell_type":"markdown","source":"在机器学习中，Landmark Indices通常指的是人脸关键点的索引。人脸关键点是人脸上的一些特定点，例如眼睛、鼻子、嘴巴等，它们可以用于人脸识别、表情识别等任务。在机器学习中，我们可以使用这些关键点来训练模型，从而实现人脸识别等任务。¹","metadata":{}},{"cell_type":"code","source":"# 定义三种数据类型：左手、姿态和右手\nUSE_TYPES = ['left_hand', 'pose', 'right_hand']\n# 定义原始数据中的起始索引\nSTART_IDX = 468\n# 定义原始数据中的嘴唇关键点索引，共40个\nLIPS_IDXS0 = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\n# 定义原始数据中的左手关键点索引，共21个\nLEFT_HAND_IDXS0 = np.arange(468,489)\n# 定义原始数据中的右手关键点索引，共21个\nRIGHT_HAND_IDXS0 = np.arange(522,543)\n# 定义原始数据中的左侧姿态关键点索引，共5个\nLEFT_POSE_IDXS0 = np.array([502, 504, 506, 508, 510])\n# 定义原始数据中的右侧姿态关键点索引，共5个\nRIGHT_POSE_IDXS0 = np.array([503, 505, 507, 509, 511])\n# 定义左手优先的关键点索引，包括嘴唇、左手和左侧姿态，共66个\nLANDMARK_IDXS_LEFT_DOMINANT0 = np.concatenate((LIPS_IDXS0, LEFT_HAND_IDXS0, LEFT_POSE_IDXS0))\n# 定义右手优先的关键点索引，包括嘴唇、右手和右侧姿态，共66个\nLANDMARK_IDXS_RIGHT_DOMINANT0 = np.concatenate((LIPS_IDXS0, RIGHT_HAND_IDXS0, RIGHT_POSE_IDXS0))\n# 定义所有手部关键点索引，包括左手和右手，共42个\nHAND_IDXS0 = np.concatenate((LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0), axis=0)\n# 定义处理后数据的列数，等于66\nN_COLS = LANDMARK_IDXS_LEFT_DOMINANT0.size\n# 定义处理后数据中的嘴唇关键点索引，从0到39\nLIPS_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LIPS_IDXS0)).squeeze()\n# 定义处理后数据中的左手关键点索引，从40到60\nLEFT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_HAND_IDXS0)).squeeze()\n# 定义处理后数据中的右手关键点索引，从40到60\nRIGHT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, RIGHT_HAND_IDXS0)).squeeze()\n# 定义处理后数据中的所有手部关键点索引，从40到81\nHAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, HAND_IDXS0)).squeeze()\n# 定义处理后数据中的姿态关键点索引，从61到65\nPOSE_IDXS = np.argwhere(np.isin(LANDMARK_IDXS_LEFT_DOMINANT0, LEFT_POSE_IDXS0)).squeeze()\n# 打印出手部关键点索引的长度和处理后数据的列数\nprint(f'# HAND_IDXS: {len(HAND_IDXS)}, N_COLS: {N_COLS}')\n#这段代码是用于定义处理后数据中不同部位的关键点的起始位置，以便于后续的切片或索引操作","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:38.238599Z","iopub.execute_input":"2023-05-01T13:31:38.239056Z","iopub.status.idle":"2023-05-01T13:31:38.253689Z","shell.execute_reply.started":"2023-05-01T13:31:38.239007Z","shell.execute_reply":"2023-05-01T13:31:38.252398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 定义嘴唇关键点的起始位置，为0\nLIPS_START = 0\n# 定义左手关键点的起始位置，为嘴唇关键点的数量\nLEFT_HAND_START = LIPS_IDXS.size\n# 定义右手关键点的起始位置，为左手关键点的起始位置加上左手关键点的数量\nRIGHT_HAND_START = LEFT_HAND_START + LEFT_HAND_IDXS.size\n# 定义姿态关键点的起始位置，为右手关键点的起始位置加上右手关键点的数量\nPOSE_START = RIGHT_HAND_START + RIGHT_HAND_IDXS.size\n# 打印出不同部位的关键点的起始位置\nprint(f'LIPS_START: {LIPS_START}, LEFT_HAND_START: {LEFT_HAND_START}, RIGHT_HAND_START: {RIGHT_HAND_START}, POSE_START: {POSE_START}')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:38.256002Z","iopub.execute_input":"2023-05-01T13:31:38.256460Z","iopub.status.idle":"2023-05-01T13:31:38.267774Z","shell.execute_reply.started":"2023-05-01T13:31:38.256423Z","shell.execute_reply":"2023-05-01T13:31:38.266642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process Data Tensorflow","metadata":{}},{"cell_type":"code","source":"# Source: https://www.kaggle.com/competitions/asl-signs/overview/evaluation\nROWS_PER_FRAME = 543  # number of landmarks per frame每帧的标记数量\n#`load_relevant_data_subset 加载相关数据子集\n#提取数据集中的x、y、z三列数据，将数据集按照每帧的地标数量进行切分，返回一个三维数组。\n#其中第一维表示帧数，第二维表示每帧的地标数量，第三维表示x、y、z三个坐标轴。\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:38.269903Z","iopub.execute_input":"2023-05-01T13:31:38.270446Z","iopub.status.idle":"2023-05-01T13:31:38.278832Z","shell.execute_reply.started":"2023-05-01T13:31:38.270253Z","shell.execute_reply":"2023-05-01T13:31:38.277792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"PreprocessLayer 类是一个继承自 TensorFlow 的 tf.keras.layers.Layer 类的自定义层，用于在 TensorFlow Lite 模型中处理数据。这个自定义层包含了一些函数，用于处理输入数据。","metadata":{}},{"cell_type":"code","source":"\"\"\"\n    Tensorflow layer to process data in TFLite\n    Data needs to be processed in the model itself, so we can not use Python\n\"\"\" \nclass PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__()\n        normalisation_correction = tf.constant([\n                    # Add 0.50 to left hand (original right hand) and substract 0.50 of right hand (original left hand)\n                    [0] * len(LIPS_IDXS) + [0.50] * len(LEFT_HAND_IDXS) + [0.50] * len(POSE_IDXS),\n                    # Y coordinates stay intact\n                    [0] * len(LANDMARK_IDXS_LEFT_DOMINANT0),\n                    # Z coordinates stay intact\n                    [0] * len(LANDMARK_IDXS_LEFT_DOMINANT0),\n                ],\n                dtype=tf.float32,\n            )\n        self.normalisation_correction = tf.transpose(normalisation_correction, [1,0])\n        \n    def pad_edge(self, t, repeats, side):\n        if side == 'LEFT':\n            return tf.concat((tf.repeat(t[:1], repeats=repeats, axis=0), t), axis=0)\n        elif side == 'RIGHT':\n            return tf.concat((t, tf.repeat(t[-1:], repeats=repeats, axis=0)), axis=0)\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,N_ROWS,N_DIMS], dtype=tf.float32),),\n    )\n    def call(self, data0):\n        # Number of Frames in Video\n        N_FRAMES0 = tf.shape(data0)[0]\n        \n        # Find dominant hand by comparing summed absolute coordinates\n        left_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0, 1))\n        right_hand_sum = tf.math.reduce_sum(tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0, 1))\n        left_dominant = left_hand_sum >= right_hand_sum\n        \n        # Count non NaN Hand values in each frame for the dominant hand\n        if left_dominant:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, LEFT_HAND_IDXS0, axis=1)), 0, 1),\n                    axis=[1, 2],\n                )\n        else:\n            frames_hands_non_nan_sum = tf.math.reduce_sum(\n                    tf.where(tf.math.is_nan(tf.gather(data0, RIGHT_HAND_IDXS0, axis=1)), 0, 1),\n                    axis=[1, 2],\n                )\n        \n        # Find frames indices with coordinates of dominant hand\n        non_empty_frames_idxs = tf.where(frames_hands_non_nan_sum > 0)\n        non_empty_frames_idxs = tf.squeeze(non_empty_frames_idxs, axis=1)\n        # Filter frames\n        data = tf.gather(data0, non_empty_frames_idxs, axis=0)\n        \n        # Cast Indices in float32 to be compatible with Tensorflow Lite\n        non_empty_frames_idxs = tf.cast(non_empty_frames_idxs, tf.float32)\n        # Normalize to start with 0\n        non_empty_frames_idxs -= tf.reduce_min(non_empty_frames_idxs)\n        \n        # Number of Frames in Filtered Video\n        N_FRAMES = tf.shape(data)[0]\n        \n        # Gather Relevant Landmark Columns\n        if left_dominant:\n            data = tf.gather(data, LANDMARK_IDXS_LEFT_DOMINANT0, axis=1)\n        else:\n            data = tf.gather(data, LANDMARK_IDXS_RIGHT_DOMINANT0, axis=1)\n            data = (\n                    self.normalisation_correction + (\n                        (data - self.normalisation_correction) * tf.where(self.normalisation_correction != 0, -1.0, 1.0))\n                )\n        \n        # Video fits in INPUT_SIZE\n        if N_FRAMES < INPUT_SIZE:\n            # Pad With -1 to indicate padding\n            non_empty_frames_idxs = tf.pad(non_empty_frames_idxs, [[0, INPUT_SIZE-N_FRAMES]], constant_values=-1)\n            # Pad Data With Zeros\n            data = tf.pad(data, [[0, INPUT_SIZE-N_FRAMES], [0,0], [0,0]], constant_values=0)\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            return data, non_empty_frames_idxs\n        # Video needs to be downsampled to INPUT_SIZE\n        else:\n            # Repeat\n            if N_FRAMES < INPUT_SIZE**2:\n                repeats = tf.math.floordiv(INPUT_SIZE * INPUT_SIZE, N_FRAMES0)\n                data = tf.repeat(data, repeats=repeats, axis=0)\n                non_empty_frames_idxs = tf.repeat(non_empty_frames_idxs, repeats=repeats, axis=0)\n\n            # Pad To Multiple Of Input Size\n            pool_size = tf.math.floordiv(len(data), INPUT_SIZE)\n            if tf.math.mod(len(data), INPUT_SIZE) > 0:\n                pool_size += 1\n\n            if pool_size == 1:\n                pad_size = (pool_size * INPUT_SIZE) - len(data)\n            else:\n                pad_size = (pool_size * INPUT_SIZE) % len(data)\n\n            # Pad Start/End with Start/End value\n            pad_left = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(INPUT_SIZE, 2)\n            pad_right = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(INPUT_SIZE, 2)\n            if tf.math.mod(pad_size, 2) > 0:\n                pad_right += 1\n\n            # Pad By Concatenating Left/Right Edge Values\n            data = self.pad_edge(data, pad_left, 'LEFT')\n            data = self.pad_edge(data, pad_right, 'RIGHT')\n\n            # Pad Non Empty Frame Indices\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_left, 'LEFT')\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_right, 'RIGHT')\n\n            # Reshape to Mean Pool\n            data = tf.reshape(data, [INPUT_SIZE, -1, N_COLS, N_DIMS])\n            non_empty_frames_idxs = tf.reshape(non_empty_frames_idxs, [INPUT_SIZE, -1])\n\n            # Mean Pool\n            data = tf.experimental.numpy.nanmean(data, axis=1)\n            non_empty_frames_idxs = tf.experimental.numpy.nanmean(non_empty_frames_idxs, axis=1)\n\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            \n            return data, non_empty_frames_idxs\n    \npreprocess_layer = PreprocessLayer()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:38.280907Z","iopub.execute_input":"2023-05-01T13:31:38.281306Z","iopub.status.idle":"2023-05-01T13:31:40.994726Z","shell.execute_reply.started":"2023-05-01T13:31:38.281268Z","shell.execute_reply":"2023-05-01T13:31:40.993248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Interpolate NaN Values","metadata":{}},{"cell_type":"markdown","source":"Interpolate NaN Values是指使用插值法填充数据中的NaN值。NaN是计算机科学中数值数据类型的一类值，表示未定义或不可表示的值。NaN是Not a Number的缩写，理解为不是一个数值。在计算机中，NaN通常用于表示无效的或未定义的操作结果，如0/0、∞-∞等1。","metadata":{}},{"cell_type":"code","source":"\"\"\"\n    face: 0:468\n    left_hand: 468:489\n    pose: 489:522\n    right_hand: 522:544\n    从file_path get data\n    第一行代码调用了load_relevant_data_subset函数，从文件路径中加载原始数据。\n    第二行代码调用了preprocess_layer函数，该函数使用Tensorflow处理数据。\n    最后返回处理后的数据。    \n\"\"\"\ndef get_data(file_path):\n    # Load Raw Data\n    data = load_relevant_data_subset(file_path)\n    # Process Data Using Tensorflow\n    data = preprocess_layer(data)\n    \n    return data","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:40.996495Z","iopub.execute_input":"2023-05-01T13:31:40.997127Z","iopub.status.idle":"2023-05-01T13:31:41.003371Z","shell.execute_reply.started":"2023-05-01T13:31:40.997088Z","shell.execute_reply":"2023-05-01T13:31:41.001983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sample = load_relevant_data_subset(train.file_path[0])\n# sample.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:41.005387Z","iopub.execute_input":"2023-05-01T13:31:41.006238Z","iopub.status.idle":"2023-05-01T13:31:41.019382Z","shell.execute_reply.started":"2023-05-01T13:31:41.006198Z","shell.execute_reply":"2023-05-01T13:31:41.018467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# preprocess_layer(sample)\n# data, non_empty_frames_idxs = preprocess_layer(sample)\n# data.shape, non_empty_frames_idxs","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:41.021279Z","iopub.execute_input":"2023-05-01T13:31:41.021643Z","iopub.status.idle":"2023-05-01T13:31:41.030067Z","shell.execute_reply.started":"2023-05-01T13:31:41.021607Z","shell.execute_reply":"2023-05-01T13:31:41.028976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # free up RAM, delete variables as we go\n# del data; del non_empty_frames_idxs","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:41.031748Z","iopub.execute_input":"2023-05-01T13:31:41.032212Z","iopub.status.idle":"2023-05-01T13:31:41.040982Z","shell.execute_reply.started":"2023-05-01T13:31:41.032177Z","shell.execute_reply":"2023-05-01T13:31:41.039877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Dataset","metadata":{}},{"cell_type":"code","source":"# Get the full dataset\ndef preprocess_data():\n    # Create arrays to save data\n    X = np.zeros([N_SAMPLES, INPUT_SIZE, N_COLS, N_DIMS], dtype=np.float32)\n    y = np.zeros([N_SAMPLES], dtype=np.int32)\n    NON_EMPTY_FRAME_IDXS = np.full([N_SAMPLES, INPUT_SIZE], -1, dtype=np.float32)\n\n    # Fill X/y\n    for row_idx, (file_path, sign_ord) in enumerate(tqdm(train[['file_path', 'sign_ord']].values)):\n        # Log message every 5000 samples\n        if row_idx % 5000 == 0:\n            print(f'Generated {row_idx}/{N_SAMPLES}')\n\n        data, non_empty_frame_idxs = get_data(file_path)\n        X[row_idx] = data\n        y[row_idx] = sign_ord\n        NON_EMPTY_FRAME_IDXS[row_idx] = non_empty_frame_idxs\n        # Sanity check, data should not contain NaN values\n        if np.isnan(data).sum() > 0:\n            print(row_idx)\n            return data\n\n    # Save X/y\n    np.save('X.npy', X)\n    np.save('y.npy', y)\n    np.save('NON_EMPTY_FRAME_IDXS.npy', NON_EMPTY_FRAME_IDXS)\n    \n    # Save Validation\n    splitter = GroupShuffleSplit(test_size=0.10, n_splits=2, random_state=SEED)\n    PARTICIPANT_IDS = train['participant_id'].values\n    train_idxs, val_idxs = next(splitter.split(X, y, groups=PARTICIPANT_IDS))\n\n    # Save Train\n    X_train = X[train_idxs]\n    NON_EMPTY_FRAME_IDXS_TRAIN = NON_EMPTY_FRAME_IDXS[train_idxs]\n    y_train = y[train_idxs]\n    np.save('X_train.npy', X_train)\n    np.save('y_train.npy', y_train)\n    np.save('NON_EMPTY_FRAME_IDXS_TRAIN.npy', NON_EMPTY_FRAME_IDXS_TRAIN)\n    # Save Validation\n    X_val = X[val_idxs]\n    NON_EMPTY_FRAME_IDXS_VAL = NON_EMPTY_FRAME_IDXS[val_idxs]\n    y_val = y[val_idxs]\n    np.save('X_val.npy', X_val)\n    np.save('y_val.npy', y_val)\n    np.save('NON_EMPTY_FRAME_IDXS_VAL.npy', NON_EMPTY_FRAME_IDXS_VAL)\n    # Split Statistics\n    print(f'Patient ID Intersection Train/Val: {set(PARTICIPANT_IDS[train_idxs]).intersection(PARTICIPANT_IDS[val_idxs])}')\n    print(f'X_train shape: {X_train.shape}, X_val shape: {X_val.shape}')\n    print(f'y_train shape: {y_train.shape}, y_val shape: {y_val.shape}')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:41.042578Z","iopub.execute_input":"2023-05-01T13:31:41.043579Z","iopub.status.idle":"2023-05-01T13:31:41.056595Z","shell.execute_reply.started":"2023-05-01T13:31:41.043550Z","shell.execute_reply":"2023-05-01T13:31:41.055838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess All Data From Scratch\nif PREPROCESS_DATA:\n    preprocess_data()\n    ROOT_DIR = '.'\nelse:\n    ROOT_DIR = '/kaggle/input/gislr-dataset-public'\n    \n# Load Data\nif USE_VAL:\n    # Load Train\n    X_train = np.load(f'{ROOT_DIR}/X_train.npy')\n    y_train = np.load(f'{ROOT_DIR}/y_train.npy')\n    NON_EMPTY_FRAME_IDXS_TRAIN = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS_TRAIN.npy')\n    # Load Val\n    X_val = np.load(f'{ROOT_DIR}/X_val.npy')\n    y_val = np.load(f'{ROOT_DIR}/y_val.npy')\n    NON_EMPTY_FRAME_IDXS_VAL = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS_VAL.npy')\n    # Define validation Data\n    validation_data = ({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, y_val)\nelse:\n    X_train = np.load(f'{ROOT_DIR}/X.npy')\n    y_train = np.load(f'{ROOT_DIR}/y.npy')\n    NON_EMPTY_FRAME_IDXS_TRAIN = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS.npy')\n    validation_data = None\n\n# Train \nprint_shape_dtype([X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN], ['X_train', 'y_train', 'NON_EMPTY_FRAME_IDXS_TRAIN'])\n# Val\nif USE_VAL:\n    print_shape_dtype([X_val, y_val, NON_EMPTY_FRAME_IDXS_VAL], ['X_val', 'y_val', 'NON_EMPTY_FRAME_IDXS_VAL'])\n# Sanity Check\nprint(f'# NaN Values X_train: {np.isnan(X_train).sum()}')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:31:41.059189Z","iopub.execute_input":"2023-05-01T13:31:41.060069Z","iopub.status.idle":"2023-05-01T13:32:20.787731Z","shell.execute_reply.started":"2023-05-01T13:31:41.060026Z","shell.execute_reply":"2023-05-01T13:32:20.786447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Vast majority of samples fits has less than 32 non empty frames\nN_EMPTY_FRAMES = (NON_EMPTY_FRAME_IDXS_TRAIN != -1).sum(axis=1) \nN_EMPTY_FRAMES_WATERFALL = []\nfor n in tqdm(range(1,INPUT_SIZE+1)):\n    N_EMPTY_FRAMES_WATERFALL.append(sum(N_EMPTY_FRAMES >= n) / len(NON_EMPTY_FRAME_IDXS_TRAIN) * 100)\n\nplt.figure(figsize=(18,10))\nplt.title('Waterfall Plot For Number Of Non Empty Frames')\npd.Series(N_EMPTY_FRAMES_WATERFALL).plot(kind='bar')\nplt.grid(axis='y')\nplt.xticks(np.arange(INPUT_SIZE), np.arange(1, INPUT_SIZE+1))\nplt.xlabel('Number of Non Empty Frames', size=16)\nplt.yticks(np.arange(0, 100+10, 10))\nplt.ylim(0, 100)\nplt.ylabel('Percentage of Samples With At Least N Non Empty Frames', size=16)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:32:20.789451Z","iopub.execute_input":"2023-05-01T13:32:20.789947Z","iopub.status.idle":"2023-05-01T13:32:34.745127Z","shell.execute_reply.started":"2023-05-01T13:32:20.789889Z","shell.execute_reply":"2023-05-01T13:32:34.743908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Percentage of frames filled, this is the maximum fill percentage of each landmark\nP_DATA_FILLED = (NON_EMPTY_FRAME_IDXS_TRAIN != -1).sum() / NON_EMPTY_FRAME_IDXS_TRAIN.size * 100\nprint(f'P_DATA_FILLED: {P_DATA_FILLED:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:32:34.746202Z","iopub.execute_input":"2023-05-01T13:32:34.746641Z","iopub.status.idle":"2023-05-01T13:32:34.762419Z","shell.execute_reply.started":"2023-05-01T13:32:34.746595Z","shell.execute_reply":"2023-05-01T13:32:34.761211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_lips_mean_std():\n    # LIPS\n    LIPS_MEAN_X = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_MEAN_Y = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_STD_X = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n    LIPS_STD_Y = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n\n    fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,LIPS_IDXS], [2,3,0,1]).reshape([LIPS_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                LIPS_MEAN_X[col] = v.mean()\n                LIPS_STD_X[col] = v.std()\n            if dim == 1: # Y\n                LIPS_MEAN_Y[col] = v.mean()\n                LIPS_STD_Y[col] = v.std()\n\n            axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    for ax, dim_name in zip(axes, DIM_NAMES):\n        ax.set_title(f'Lips {dim_name.upper()} Dimension', size=24)\n        ax.tick_params(axis='x', labelsize=8)\n        ax.grid(axis='y')\n\n    plt.subplots_adjust(hspace=0.50)\n    plt.show()\n\n    LIPS_MEAN = np.array([LIPS_MEAN_X, LIPS_MEAN_Y]).T\n    LIPS_STD = np.array([LIPS_STD_X, LIPS_STD_Y]).T\n    \n    return LIPS_MEAN, LIPS_STD\n\nLIPS_MEAN, LIPS_STD = get_lips_mean_std()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:32:34.771864Z","iopub.execute_input":"2023-05-01T13:32:34.772304Z","iopub.status.idle":"2023-05-01T13:32:56.288453Z","shell.execute_reply.started":"2023-05-01T13:32:34.772275Z","shell.execute_reply":"2023-05-01T13:32:56.287281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Verify Normalised to Left Hand Dominant\nP_LEFT_HAND_MEASUREMENTS = (X_train[:,:,LEFT_HAND_IDXS] != 0).sum() / X_train[:,:,LEFT_HAND_IDXS].size / P_DATA_FILLED * 1e4\n# P_RIGHT_HAND_MEASUREMENTS = (X_train[:,:,RIGHT_HAND_IDXS] != 0).sum() / X_train[:,:,RIGHT_HAND_IDXS].size / P_DATA_FILLED * 1e4\nprint(f'P_LEFT_HAND_MEASUREMENTS: {P_LEFT_HAND_MEASUREMENTS:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:32:56.290138Z","iopub.execute_input":"2023-05-01T13:32:56.290528Z","iopub.status.idle":"2023-05-01T13:33:05.972791Z","shell.execute_reply.started":"2023-05-01T13:32:56.290491Z","shell.execute_reply":"2023-05-01T13:33:05.971496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_left_right_hand_mean_std():\n    # LEFT HAND\n    LEFT_HANDS_MEAN_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_MEAN_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_STD_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n    LEFT_HANDS_STD_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n\n    fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,LEFT_HAND_IDXS], [2,3,0,1]).reshape([LEFT_HAND_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                LEFT_HANDS_MEAN_X[col] = v.mean()\n                LEFT_HANDS_STD_X[col] = v.std()\n            if dim == 1: # Y\n                LEFT_HANDS_MEAN_Y[col] = v.mean()\n                LEFT_HANDS_STD_Y[col] = v.std()\n            # Plot\n            axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    for ax, dim_name in zip(axes, DIM_NAMES):\n        ax.set_title(f'Hands {dim_name.upper()} Dimension', size=24)\n        ax.tick_params(axis='x', labelsize=8)\n        ax.grid(axis='y')\n\n    plt.subplots_adjust(hspace=0.50)\n    plt.show()\n\n    LEFT_HANDS_MEAN = np.array([LEFT_HANDS_MEAN_X, LEFT_HANDS_MEAN_Y]).T\n    LEFT_HANDS_STD = np.array([LEFT_HANDS_STD_X, LEFT_HANDS_STD_Y]).T\n    \n    return LEFT_HANDS_MEAN, LEFT_HANDS_STD\n\nLEFT_HANDS_MEAN, LEFT_HANDS_STD = get_left_right_hand_mean_std()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:05.974216Z","iopub.execute_input":"2023-05-01T13:33:05.975125Z","iopub.status.idle":"2023-05-01T13:33:17.256862Z","shell.execute_reply.started":"2023-05-01T13:33:05.975093Z","shell.execute_reply":"2023-05-01T13:33:17.255869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Percentage of Lips Measurements\nP_POSE_MEASUREMENTS = (X_train[:,:,POSE_IDXS] != 0).sum() / X_train[:,:,POSE_IDXS].size / P_DATA_FILLED * 1e4\nprint(f'P_POSE_MEASUREMENTS: {P_POSE_MEASUREMENTS:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:17.258703Z","iopub.execute_input":"2023-05-01T13:33:17.259459Z","iopub.status.idle":"2023-05-01T13:33:19.123563Z","shell.execute_reply.started":"2023-05-01T13:33:17.259419Z","shell.execute_reply":"2023-05-01T13:33:19.122301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_pose_mean_std():\n    # POSE\n    POSE_MEAN_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_MEAN_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_STD_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\n    POSE_STD_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\n\n    fig, axes = plt.subplots(3, 1, figsize=(15, N_DIMS*6))\n\n    for col, ll in enumerate(tqdm( np.transpose(X_train[:,:,POSE_IDXS], [2,3,0,1]).reshape([POSE_IDXS.size, N_DIMS, -1]) )):\n        for dim, l in enumerate(ll):\n            v = l[np.nonzero(l)]\n            if dim == 0: # X\n                POSE_MEAN_X[col] = v.mean()\n                POSE_STD_X[col] = v.std()\n            if dim == 1: # Y\n                POSE_MEAN_Y[col] = v.mean()\n                POSE_STD_Y[col] = v.std()\n\n            axes[dim].boxplot(v, notch=False, showfliers=False, positions=[col], whis=[5,95])\n\n    for ax, dim_name in zip(axes, DIM_NAMES):\n        ax.set_title(f'Pose {dim_name.upper()} Dimension', size=24)\n        ax.tick_params(axis='x', labelsize=8)\n        ax.grid(axis='y')\n\n    plt.subplots_adjust(hspace=0.50)\n    plt.show()\n\n    POSE_MEAN = np.array([POSE_MEAN_X, POSE_MEAN_Y]).T\n    POSE_STD = np.array([POSE_STD_X, POSE_STD_Y]).T\n    \n    return POSE_MEAN, POSE_STD\n\nPOSE_MEAN, POSE_STD = get_pose_mean_std()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:19.126781Z","iopub.execute_input":"2023-05-01T13:33:19.127079Z","iopub.status.idle":"2023-05-01T13:33:22.077875Z","shell.execute_reply.started":"2023-05-01T13:33:19.127052Z","shell.execute_reply":"2023-05-01T13:33:22.076727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Custom sampler to get a batch containing N times all signs\ndef get_train_batch_all_signs(X, y, NON_EMPTY_FRAME_IDXS, n=BATCH_ALL_SIGNS_N):\n    # Arrays to store batch in\n    X_batch = np.zeros([NUM_CLASSES*n, INPUT_SIZE, N_COLS, N_DIMS], dtype=np.float32)\n    y_batch = np.arange(0, NUM_CLASSES, step=1/n, dtype=np.float32).astype(np.int64)\n    non_empty_frame_idxs_batch = np.zeros([NUM_CLASSES*n, INPUT_SIZE], dtype=np.float32)\n    \n    # Dictionary mapping ordinally encoded sign to corresponding sample indices\n    CLASS2IDXS = {}\n    for i in range(NUM_CLASSES):\n        CLASS2IDXS[i] = np.argwhere(y == i).squeeze().astype(np.int32)\n            \n    while True:\n        # Fill batch arrays\n        for i in range(NUM_CLASSES):\n            idxs = np.random.choice(CLASS2IDXS[i], n)\n            X_batch[i*n:(i+1)*n] = X[idxs]\n            non_empty_frame_idxs_batch[i*n:(i+1)*n] = NON_EMPTY_FRAME_IDXS[idxs]\n        \n        yield { 'frames': X_batch, 'non_empty_frame_idxs': non_empty_frame_idxs_batch }, y_batch","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:22.079753Z","iopub.execute_input":"2023-05-01T13:33:22.080169Z","iopub.status.idle":"2023-05-01T13:33:22.090312Z","shell.execute_reply.started":"2023-05-01T13:33:22.080130Z","shell.execute_reply":"2023-05-01T13:33:22.089182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dummy_dataset = get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN)\nX_batch, y_batch = next(dummy_dataset)\n\nfor k, v in X_batch.items():\n    print(f'{k} shape: {v.shape}, dtype: {v.dtype}')\n\n# Batch shape/dtype\nprint(f'y_batch shape: {y_batch.shape}, dtype: {y_batch.dtype}')\n# Verify each batch contains each sign exactly N times\ndisplay(pd.Series(y_batch).value_counts().to_frame('Counts'))","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:22.092025Z","iopub.execute_input":"2023-05-01T13:33:22.092475Z","iopub.status.idle":"2023-05-01T13:33:22.179218Z","shell.execute_reply.started":"2023-05-01T13:33:22.092437Z","shell.execute_reply":"2023-05-01T13:33:22.177980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Epsilon value for layer normalisation\n#epsilon 值在机器学习中，层归一化是一种归一化技术，用于在神经网络的每个层中标准化输入。这有助于加速训练并提高模型的准确性。\n#在层归一化中，epsilon 值是一个常量，用于设置层归一化的 epsilon 值。它是一个非常小的数，通常设置为 1e-5 或 1e-6。\n#它的作用是防止分母为零，从而避免数值计算不稳定。\n\n\nLAYER_NORM_EPS = 1e-6\n\n# Dense layer units for landmarks\nLIPS_UNITS = 384\nHANDS_UNITS = 384\nPOSE_UNITS = 384\n# final embedding and transformer embedding size\nUNITS = 512\n\n# Transformer\n#mlp_ratio代表第一个全连接层上升通道倍数；\nNUM_BLOCKS = 2\nMLP_RATIO = 4\n\n# Dropout\n#模型很重要的性质就是非线性，\n#同时为了模型泛化能力，需要加入随机正则，例如dropout(随机置一些输出为0,其实也是一种变相的随机非线性激活)\nEMBEDDING_DROPOUT = 0.00\nMLP_DROPOUT_RATIO = 0.15\nCLASSIFIER_DROPOUT_RATIO = 0.00\n\n# Initiailizers\nINIT_HE_UNIFORM = tf.keras.initializers.he_uniform\nINIT_GLOROT_UNIFORM = tf.keras.initializers.glorot_uniform\nINIT_ZEROS = tf.keras.initializers.constant(0.0)\n# Activations\nGELU = tf.keras.activations.gelu\n\nprint(f'UNITS: {UNITS}')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:22.180753Z","iopub.execute_input":"2023-05-01T13:33:22.181375Z","iopub.status.idle":"2023-05-01T13:33:22.189689Z","shell.execute_reply.started":"2023-05-01T13:33:22.181334Z","shell.execute_reply":"2023-05-01T13:33:22.188310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DROP_Z = True\n# NUM_FRAMES = 15\n# SEGMENTS = 3\n\n# LEFT_HAND_OFFSET = 468\n# POSE_OFFSET = LEFT_HAND_OFFSET+21\n# RIGHT_HAND_OFFSET = POSE_OFFSET+33\n\n# ## average over the entire face, and the entire 'pose'\n# averaging_sets = [[0, 468], [POSE_OFFSET, 33]]\n\n# lip_landmarks = [61, 185, 40, 39, 37,  0, 267, 269, 270, 409,\n#                  291,146, 91,181, 84, 17, 314, 405, 321, 375, \n#                  78, 191, 80, 81, 82, 13, 312, 311, 310, 415, \n#                  95, 88, 178, 87, 14,317, 402, 318, 324, 308]\n# left_hand_landmarks = list(range(LEFT_HAND_OFFSET, LEFT_HAND_OFFSET+21))\n# LEFT_POSE_IDXS0 = np.array([502, 504, 506, 508, 510])\n# right_hand_landmarks = list(range(RIGHT_HAND_OFFSET, RIGHT_HAND_OFFSET+21))\n# HAND_IDXS0 = np.concatenate((left_hand_landmarks, right_hand_landmarks), axis=0)\n# point_landmarks = np.concatenate((lip_landmarks, left_hand_landmarks, right_hand_landmarks), axis=0)\n\n\n# LANDMARKS = len(point_landmarks) + len(averaging_sets)\n# print(LANDMARKS)\n# INPUT_SHAPE = (NUM_FRAMES,LANDMARKS*2)\n\n# FLAT_INPUT_SHAPE = (INPUT_SHAPE[0] + 2 * (SEGMENTS + 1)) * INPUT_SHAPE[1]\n# print(INPUT_SHAPE, FLAT_INPUT_SHAPE)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:22.191283Z","iopub.execute_input":"2023-05-01T13:33:22.191769Z","iopub.status.idle":"2023-05-01T13:33:22.204359Z","shell.execute_reply.started":"2023-05-01T13:33:22.191733Z","shell.execute_reply":"2023-05-01T13:33:22.203169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def tf_nan_mean(x, axis=0):\n#     return tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), x), axis=axis) / tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), tf.ones_like(x)), axis=axis)\n\n# def tf_nan_std(x, axis=0):\n#     d = x - tf_nan_mean(x, axis=axis)\n#     return tf.math.sqrt(tf_nan_mean(d * d, axis=axis))\n\n# def flatten_means_and_stds(x, axis=0):\n#     # Get means and stds\n#     x_mean = tf_nan_mean(x, axis=0)\n#     x_std  = tf_nan_std(x,  axis=0)\n\n#     x_out = tf.concat([x_mean, x_std], axis=0)\n#     x_out = tf.reshape(x_out, (1, INPUT_SHAPE[1]*2))\n#     x_out = tf.where(tf.math.is_finite(x_out), x_out, tf.zeros_like(x_out))\n#     return x_out\n\n# class FeatureGen(tf.keras.layers.Layer):\n#     def __init__(self):\n#         super(FeatureGen, self).__init__()\n    \n#     def call(self, x_in):\n# #         print(right_hand_percentage(x))\n\n#         if not isinstance(x_in, (np.ndarray, tf.Tensor)): \n#             x_in = load_relevant_data_subset(x_in)\n#         if DROP_Z:\n#             x_in=x_in[:,:,0:2]\n#         x_list = [tf.expand_dims(tf_nan_mean(x_in[:, av_set[0]:av_set[0]+av_set[1], :], axis=1), axis=1) for av_set in averaging_sets]\n#         #print(len(x_list))\n#         #print(len(point_landmarks))\n#         x_list.append(tf.gather(x_in, point_landmarks, axis=1))\n#         #print(x_list)\n#         x = tf.concat(x_list, 1)\n#         #print(len(x[1]))\n\n#         x_padded = x\n#         #print(tf.shape(x_padded)[0])\n#         for i in range(SEGMENTS):\n#             p0 = tf.where( ((tf.shape(x_padded)[0] % SEGMENTS) > 0) & ((i % 2) != 0) , 1, 0)\n#             p1 = tf.where( ((tf.shape(x_padded)[0] % SEGMENTS) > 0) & ((i % 2) == 0) , 1, 0)\n#             paddings = [[p0, p1], [0, 0], [0,0]]\n#             #print(paddings[0])\n#             x_padded = tf.pad(x_padded, paddings, mode=\"SYMMETRIC\")\n#         x_list = tf.split(x_padded, SEGMENTS)\n#         #print(x_list[:1])\n#         x_list = [flatten_means_and_stds(_x, axis=0) for _x in x_list]\n\n#         x_list.append(flatten_means_and_stds(x, axis=0))\n        \n#         ## Resize only dimension 0. Resize can't handle nan, so replace nan with that dimension's avg value to reduce impact.\n#         x = tf.image.resize(tf.where(tf.math.is_finite(x), x, tf_nan_mean(x, axis=0)), [NUM_FRAMES, LANDMARKS])\n#         x = tf.reshape(x, (1, INPUT_SHAPE[0]*INPUT_SHAPE[1]))\n#         x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n#         x_list.append(x)\n#         x = tf.concat(x_list, axis=1)\n#         return x\n\n# feature_converter = FeatureGen()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:22.206481Z","iopub.execute_input":"2023-05-01T13:33:22.206783Z","iopub.status.idle":"2023-05-01T13:33:22.219034Z","shell.execute_reply.started":"2023-05-01T13:33:22.206746Z","shell.execute_reply":"2023-05-01T13:33:22.217939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# based on: https://stackoverflow.com/\n#questions/67342988/verifying-the-implementation-of-multihead-attention-in-transformer\n# replaced softmax with softmax layer to support masked softmax\n#scaled dot-product attention是Transformer模型中的一种Attention机制，它是一种计算Attention权重的方法。\n#在这种方法中，Query和Key的点积被除以一个缩放因子，然后通过softmax函数进行归一化处理，最后与Value相乘得到Attention输出\ndef scaled_dot_product(q,k,v, softmax, attention_mask):\n    #calculates Q . K(transpose)\n    qkt = tf.matmul(q,k,transpose_b=True)\n    #caculates scaling factor\n    dk = tf.math.sqrt(tf.cast(q.shape[-1],dtype=tf.float32))\n    scaled_qkt = qkt/dk\n    softmax = softmax(scaled_qkt, mask=attention_mask)\n    \n    z = tf.matmul(softmax,v)\n    #shape: (m,Tx,depth), same shape as q,k,v\n    return z\n\nclass MultiHeadAttention(tf.keras.layers.Layer):\n    def __init__(self,d_model,num_of_heads):\n        super(MultiHeadAttention,self).__init__()\n        self.d_model = d_model\n        self.num_of_heads = num_of_heads\n        self.depth = d_model//num_of_heads\n        self.wq = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wk = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wv = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wo = tf.keras.layers.Dense(d_model)\n        self.softmax = tf.keras.layers.Softmax()\n        \n    def call(self,x, attention_mask):\n        \n        multi_attn = []\n        for i in range(self.num_of_heads):\n            Q = self.wq[i](x)\n            K = self.wk[i](x)\n            V = self.wv[i](x)\n            multi_attn.append(scaled_dot_product(Q,K,V, self.softmax, attention_mask))\n            \n        multi_head = tf.concat(multi_attn,axis=-1)\n        multi_head_attention = self.wo(multi_head)\n        return multi_head_attention","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:22.220553Z","iopub.execute_input":"2023-05-01T13:33:22.221116Z","iopub.status.idle":"2023-05-01T13:33:22.234936Z","shell.execute_reply.started":"2023-05-01T13:33:22.221078Z","shell.execute_reply":"2023-05-01T13:33:22.233866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Full Transformer\nclass Transformer(tf.keras.Model):\n    def __init__(self, num_blocks):\n        super(Transformer, self).__init__(name='transformer')\n        self.num_blocks = num_blocks\n    \n    def build(self, input_shape):\n        self.ln_1s = []\n        self.mhas = []\n        self.ln_2s = []\n        self.mlps = []\n        # Make Transformer Blocks\n        for i in range(self.num_blocks):\n            # Multi Head Attention\n            self.mhas.append(MultiHeadAttention(UNITS, 8))\n            # Multi Layer Perception\n            self.mlps.append(tf.keras.Sequential([\n                tf.keras.layers.Dense(UNITS * MLP_RATIO, activation=GELU, kernel_initializer=INIT_GLOROT_UNIFORM),\n                tf.keras.layers.Dropout(MLP_DROPOUT_RATIO),\n                tf.keras.layers.Dense(UNITS, kernel_initializer=INIT_HE_UNIFORM),\n            ]))\n        \n    def call(self, x, attention_mask):\n        # Iterate input over transformer blocks\n        for mha, mlp in zip(self.mhas, self.mlps):\n            x = x + mha(x, attention_mask)\n            x = x + mlp(x)\n    \n        return x","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:22.237474Z","iopub.execute_input":"2023-05-01T13:33:22.237737Z","iopub.status.idle":"2023-05-01T13:33:22.250822Z","shell.execute_reply.started":"2023-05-01T13:33:22.237711Z","shell.execute_reply":"2023-05-01T13:33:22.249835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class LandmarkEmbedding(tf.keras.Model):\n    def __init__(self, units, name):\n        super(LandmarkEmbedding, self).__init__(name=f'{name}_embedding')\n        self.units = units\n        \n    def build(self, input_shape):\n        # Embedding for missing landmark in frame, initizlied with zeros\n        self.empty_embedding = self.add_weight(\n            name=f'{self.name}_empty_embedding',\n            shape=[self.units],\n            initializer=INIT_ZEROS,\n        )\n        # Embedding\n        self.dense = tf.keras.Sequential([\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM),\n            tf.keras.layers.Activation(GELU),\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name=f'{self.name}_dense')\n\n    def call(self, x):\n        return tf.where(\n                # Checks whether landmark is missing in frame\n                tf.reduce_sum(x, axis=2, keepdims=True) == 0,\n                # If so, the empty embedding is used\n                self.empty_embedding,\n                # Otherwise the landmark data is embedded\n                self.dense(x),\n            )","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:22.252413Z","iopub.execute_input":"2023-05-01T13:33:22.252829Z","iopub.status.idle":"2023-05-01T13:33:22.265359Z","shell.execute_reply.started":"2023-05-01T13:33:22.252794Z","shell.execute_reply":"2023-05-01T13:33:22.264409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Embedding(tf.keras.Model):\n    def __init__(self):\n        super(Embedding, self).__init__()\n        \n    def get_diffs(self, l):\n        S = l.shape[2]\n        other = tf.expand_dims(l, 3)\n        other = tf.repeat(other, S, axis=3)\n        other = tf.transpose(other, [0,1,3,2])\n        diffs = tf.expand_dims(l, 3) - other\n        diffs = tf.reshape(diffs, [-1, INPUT_SIZE, S*S])\n        return diffs\n\n    def build(self, input_shape):\n        # Positional Embedding, initialized with zeros\n        self.positional_embedding = tf.keras.layers.Embedding(INPUT_SIZE+1, UNITS, embeddings_initializer=INIT_ZEROS)\n        # Embedding layer for Landmarks\n        self.lips_embedding = LandmarkEmbedding(LIPS_UNITS, 'lips')\n        self.left_hand_embedding = LandmarkEmbedding(HANDS_UNITS, 'left_hand')\n        self.pose_embedding = LandmarkEmbedding(POSE_UNITS, 'pose')\n        # Landmark Weights\n        self.landmark_weights = tf.Variable(tf.zeros([3], dtype=tf.float32), name='landmark_weights')\n        # Fully Connected Layers for combined landmarks\n        self.fc = tf.keras.Sequential([\n            tf.keras.layers.Dense(UNITS, name='fully_connected_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM),\n            tf.keras.layers.Activation(GELU),\n            tf.keras.layers.Dense(UNITS, name='fully_connected_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name='fc')\n\n\n    def call(self, lips0, left_hand0, pose0, non_empty_frame_idxs, training=False):\n        # Lips\n        lips_embedding = self.lips_embedding(lips0)\n        # Left Hand\n        left_hand_embedding = self.left_hand_embedding(left_hand0)\n        # Pose\n        pose_embedding = self.pose_embedding(pose0)\n        # Merge Embeddings of all landmarks with mean pooling\n        x = tf.stack((\n            lips_embedding, left_hand_embedding, pose_embedding,\n        ), axis=3)\n        x = x * tf.nn.softmax(self.landmark_weights)\n        x = tf.reduce_sum(x, axis=3)\n        # Fully Connected Layers\n        x = self.fc(x)\n        # Add Positional Embedding\n        max_frame_idxs = tf.clip_by_value(\n                tf.reduce_max(non_empty_frame_idxs, axis=1, keepdims=True),\n                1,\n                np.PINF,\n            )\n        normalised_non_empty_frame_idxs = tf.where(\n            tf.math.equal(non_empty_frame_idxs, -1.0),\n            INPUT_SIZE,\n            tf.cast(\n                non_empty_frame_idxs / max_frame_idxs * INPUT_SIZE,\n                tf.int32,\n            ),\n        )\n        x = x + self.positional_embedding(normalised_non_empty_frame_idxs)\n        \n        return x","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:22.266965Z","iopub.execute_input":"2023-05-01T13:33:22.267514Z","iopub.status.idle":"2023-05-01T13:33:22.283670Z","shell.execute_reply.started":"2023-05-01T13:33:22.267478Z","shell.execute_reply":"2023-05-01T13:33:22.282559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# source:: https://stackoverflow.com/questions/60689185/label-smoothing-for-sparse-categorical-crossentropy\ndef scce_with_ls(y_true, y_pred):\n    # One Hot Encode Sparsely Encoded Target Sign\n    y_true = tf.cast(y_true, tf.int32)\n    y_true = tf.one_hot(y_true, NUM_CLASSES, axis=1)\n    y_true = tf.squeeze(y_true, axis=2)\n    # Categorical Crossentropy with native label smoothing support\n    return tf.keras.losses.categorical_crossentropy(y_true, y_pred, label_smoothing=0.25)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:22.285182Z","iopub.execute_input":"2023-05-01T13:33:22.286253Z","iopub.status.idle":"2023-05-01T13:33:22.297528Z","shell.execute_reply.started":"2023-05-01T13:33:22.286213Z","shell.execute_reply":"2023-05-01T13:33:22.296783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    # Inputs\n    frames = tf.keras.layers.Input([INPUT_SIZE, N_COLS, N_DIMS], dtype=tf.float32, name='frames')\n    non_empty_frame_idxs = tf.keras.layers.Input([INPUT_SIZE], dtype=tf.float32, name='non_empty_frame_idxs')\n    # Padding Mask\n    mask0 = tf.cast(tf.math.not_equal(non_empty_frame_idxs, -1), tf.float32)\n    mask0 = tf.expand_dims(mask0, axis=2)\n    # Random Frame Masking\n    mask = tf.where(\n        (tf.random.uniform(tf.shape(mask0)) > 0.25) & tf.math.not_equal(mask0, 0.0),\n        1.0,\n        0.0,\n    )\n    # Correct Samples Which are all masked now...\n    mask = tf.where(\n        tf.math.equal(tf.reduce_sum(mask, axis=[1,2], keepdims=True), 0.0),\n        mask0,\n        mask,\n    )\n    \n    \n    \"\"\"\n        left_hand: 468:489\n        pose: 489:522\n        right_hand: 522:543\n    \"\"\"\n    x = frames\n    x = tf.slice(x, [0,0,0,0], [-1,INPUT_SIZE, N_COLS, 2])\n    # LIPS\n    lips = tf.slice(x, [0,0,LIPS_START,0], [-1,INPUT_SIZE, 40, 2])\n    lips = tf.where(\n            tf.math.equal(lips, 0.0),\n            0.0,\n            (lips - LIPS_MEAN) / LIPS_STD,\n        )\n    # LEFT HAND\n    left_hand = tf.slice(x, [0,0,40,0], [-1,INPUT_SIZE, 21, 2])\n    left_hand = tf.where(\n            tf.math.equal(left_hand, 0.0),\n            0.0,\n            (left_hand - LEFT_HANDS_MEAN) / LEFT_HANDS_STD,\n        )\n    # POSE\n    pose = tf.slice(x, [0,0,61,0], [-1,INPUT_SIZE, 5, 2])\n    pose = tf.where(\n            tf.math.equal(pose, 0.0),\n            0.0,\n            (pose - POSE_MEAN) / POSE_STD,\n        )\n    \n    # Flatten\n    lips = tf.reshape(lips, [-1, INPUT_SIZE, 40*2])\n    left_hand = tf.reshape(left_hand, [-1, INPUT_SIZE, 21*2])\n    pose = tf.reshape(pose, [-1, INPUT_SIZE, 5*2])\n        \n    # Embedding\n    x = Embedding()(lips, left_hand, pose, non_empty_frame_idxs)\n    \n    # Encoder Transformer Blocks\n    x = Transformer(NUM_BLOCKS)(x, mask)\n    \n    # Pooling\n    x = tf.reduce_sum(x * mask, axis=1) / tf.reduce_sum(mask, axis=1)\n    # Classifier Dropout\n    x = tf.keras.layers.Dropout(CLASSIFIER_DROPOUT_RATIO)(x)\n    # Classification Layer\n    x = tf.keras.layers.Dense(NUM_CLASSES, activation=tf.keras.activations.softmax, kernel_initializer=INIT_GLOROT_UNIFORM)(x)\n    \n    outputs = x\n    \n    # Create Tensorflow Model\n    model = tf.keras.models.Model(inputs=[frames, non_empty_frame_idxs], outputs=outputs)\n    \n    # Sparse Categorical Cross Entropy With Label Smoothing\n    loss = scce_with_ls\n    #SGDW是一种优化器，它是基于SGD的，但是加入了动量的概念。\n    #动量的作用是在更新参数时，不仅仅减去了当前迭代的梯度，还减去了前t-1迭代的梯度的加权和。\n    #这样做的好处是可以让参数更新更加平滑，避免了在参数更新过程中出现震荡的情况。\n    #optimizer = tfa.optimizers.SGDW(\n    #learning_rate=lr, weight_decay=wd, momentum=0.9)\n    #optimizer = tf.keras.optimizers.SGD(lr=0.001, momentum=0.0, nesterov=False) \n    #optimizer = tfa.optimizers.SGDW(learning_rate=0.001, momentum=0.7, weight_decay=0.005)\n    #Adam Optimizer with weight decay\n    optimizer = tfa.optimizers.AdamW(learning_rate=1e-3, weight_decay=1e-5, clipnorm=1.0)\n    #学习率为1e-3，权重衰减为1e-5，梯度裁剪阈值为1.0\n    # TopK Metrics\n    metrics = [\n        tf.keras.metrics.SparseCategoricalAccuracy(name='acc'),\n        tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5, name='top_5_acc'),\n        tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10, name='top_10_acc'),\n    ]\n    \n    model.compile(loss=loss, optimizer=optimizer, metrics=metrics)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:22.299598Z","iopub.execute_input":"2023-05-01T13:33:22.300049Z","iopub.status.idle":"2023-05-01T13:33:22.317800Z","shell.execute_reply.started":"2023-05-01T13:33:22.300012Z","shell.execute_reply":"2023-05-01T13:33:22.316793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nmodel = get_model()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:22.319351Z","iopub.execute_input":"2023-05-01T13:33:22.320356Z","iopub.status.idle":"2023-05-01T13:33:24.456299Z","shell.execute_reply.started":"2023-05-01T13:33:22.320317Z","shell.execute_reply":"2023-05-01T13:33:24.455279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot model summary\nmodel.summary(expand_nested=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:24.458002Z","iopub.execute_input":"2023-05-01T13:33:24.458719Z","iopub.status.idle":"2023-05-01T13:33:24.610931Z","shell.execute_reply.started":"2023-05-01T13:33:24.458675Z","shell.execute_reply":"2023-05-01T13:33:24.610125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(model, show_shapes=True, show_dtype=True, show_layer_names=True, expand_nested=True, show_layer_activations=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:24.612000Z","iopub.execute_input":"2023-05-01T13:33:24.612345Z","iopub.status.idle":"2023-05-01T13:33:25.634457Z","shell.execute_reply.started":"2023-05-01T13:33:24.612306Z","shell.execute_reply":"2023-05-01T13:33:25.633183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not PREPROCESS_DATA and TRAIN_MODEL:\n    y_pred = model.predict_on_batch(X_batch).flatten()\n\n    print(f'# NaN Values In Prediction: {np.isnan(y_pred).sum()}')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:25.636735Z","iopub.execute_input":"2023-05-01T13:33:25.637453Z","iopub.status.idle":"2023-05-01T13:33:29.514086Z","shell.execute_reply.started":"2023-05-01T13:33:25.637412Z","shell.execute_reply":"2023-05-01T13:33:29.512604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not PREPROCESS_DATA and TRAIN_MODEL:\n    plt.figure(figsize=(12,5))\n    plt.title(f'Softmax Output Initialized Model | µ={y_pred.mean():.3f}, σ={y_pred.std():.3f}', pad=25)\n    pd.Series(y_pred).plot(kind='hist', bins=128, label='Class Probability')\n    plt.xlim(0, max(y_pred) * 1.1)\n    plt.vlines([1 / NUM_CLASSES], 0, plt.ylim()[1], color='red', label=f'Random Guessing Baseline 1/NUM_CLASSES={1 / NUM_CLASSES:.3f}')\n    plt.grid()\n    plt.legend()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:29.515868Z","iopub.execute_input":"2023-05-01T13:33:29.516563Z","iopub.status.idle":"2023-05-01T13:33:30.329398Z","shell.execute_reply.started":"2023-05-01T13:33:29.516518Z","shell.execute_reply":"2023-05-01T13:33:30.328303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lrfn(current_step, num_warmup_steps, lr_max, num_cycles=0.50, num_training_steps=N_EPOCHS):\n    \n    if current_step < num_warmup_steps:\n        if WARMUP_METHOD == 'log':\n            return lr_max * 0.10 ** (num_warmup_steps - current_step)\n        else:\n            return lr_max * 2 ** -(num_warmup_steps - current_step)\n    else:\n        progress = float(current_step - num_warmup_steps) / float(max(1, num_training_steps - num_warmup_steps))\n\n        return max(0.0, 0.5 * (1.0 + math.cos(math.pi * float(num_cycles) * 2.0 * progress))) * lr_max","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:30.334299Z","iopub.execute_input":"2023-05-01T13:33:30.336960Z","iopub.status.idle":"2023-05-01T13:33:30.346689Z","shell.execute_reply.started":"2023-05-01T13:33:30.336903Z","shell.execute_reply":"2023-05-01T13:33:30.345739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_lr_schedule(lr_schedule, epochs):\n    fig = plt.figure(figsize=(20, 10))\n    plt.plot([None] + lr_schedule + [None])\n    # X Labels\n    x = np.arange(1, epochs + 1)\n    x_axis_labels = [i if epochs <= 40 or i % 5 == 0 or i == 1 else None for i in range(1, epochs + 1)]\n    plt.xlim([1, epochs])\n    plt.xticks(x, x_axis_labels) # set tick step to 1 and let x axis start at 1\n    \n    # Increase y-limit for better readability\n    plt.ylim([0, max(lr_schedule) * 1.1])\n    \n    # Title\n    schedule_info = f'start: {lr_schedule[0]:.1E}, max: {max(lr_schedule):.1E}, final: {lr_schedule[-1]:.1E}'\n    plt.title(f'Step Learning Rate Schedule, {schedule_info}', size=18, pad=12)\n    \n    # Plot Learning Rates\n    for x, val in enumerate(lr_schedule):\n        if epochs <= 40 or x % 5 == 0 or x is epochs - 1:\n            if x < len(lr_schedule) - 1:\n                if lr_schedule[x - 1] < val:\n                    ha = 'right'\n                else:\n                    ha = 'left'\n            elif x == 0:\n                ha = 'right'\n            else:\n                ha = 'left'\n            plt.plot(x + 1, val, 'o', color='black');\n            offset_y = (max(lr_schedule) - min(lr_schedule)) * 0.02\n            plt.annotate(f'{val:.1E}', xy=(x + 1, val + offset_y), size=12, ha=ha)\n    \n    plt.xlabel('Epoch', size=16, labelpad=5)\n    plt.ylabel('Learning Rate', size=16, labelpad=5)\n    plt.grid()\n    plt.show()\n\n# Learning rate for encoder\nLR_SCHEDULE = [lrfn(step, num_warmup_steps=N_WARMUP_EPOCHS, lr_max=LR_MAX, num_cycles=0.50) for step in range(N_EPOCHS)]\n# Plot Learning Rate Schedule\nplot_lr_schedule(LR_SCHEDULE, epochs=N_EPOCHS)\n# Learning Rate Callback\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lambda step: LR_SCHEDULE[step], verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:30.351816Z","iopub.execute_input":"2023-05-01T13:33:30.356171Z","iopub.status.idle":"2023-05-01T13:33:30.641142Z","shell.execute_reply.started":"2023-05-01T13:33:30.356075Z","shell.execute_reply":"2023-05-01T13:33:30.640007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Custom callback to update weight decay with learning rate\nclass WeightDecayCallback(tf.keras.callbacks.Callback):\n    def __init__(self, wd_ratio=WD_RATIO):\n        self.step_counter = 0\n        self.wd_ratio = wd_ratio\n    \n    def on_epoch_begin(self, epoch, logs=None):\n        model.optimizer.weight_decay = model.optimizer.learning_rate * self.wd_ratio\n        print(f'learning rate: {model.optimizer.learning_rate.numpy():.2e}, weight decay: {model.optimizer.weight_decay.numpy():.2e}')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:30.642720Z","iopub.execute_input":"2023-05-01T13:33:30.644144Z","iopub.status.idle":"2023-05-01T13:33:30.651438Z","shell.execute_reply.started":"2023-05-01T13:33:30.644093Z","shell.execute_reply":"2023-05-01T13:33:30.650355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%timeit -n 100\nif TRAIN_MODEL:\n    # Verify model prediction is <<<100ms\n    model.predict_on_batch({ 'frames': X_train[:1], 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_TRAIN[:1] })\n    pass","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:30.654815Z","iopub.execute_input":"2023-05-01T13:33:30.655935Z","iopub.status.idle":"2023-05-01T13:33:43.444713Z","shell.execute_reply.started":"2023-05-01T13:33:30.655886Z","shell.execute_reply":"2023-05-01T13:33:43.443624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if USE_VAL:\n    # Verify Validation Dataset Covers All Signs\n    print(f'# Unique Signs in Validation Set: {pd.Series(y_val).nunique()}')\n    # Value Counts\n    display(pd.Series(y_val).value_counts().to_frame('Count').iloc[[1,2,3,-3,-2,-1]])","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:43.447075Z","iopub.execute_input":"2023-05-01T13:33:43.447474Z","iopub.status.idle":"2023-05-01T13:33:43.453605Z","shell.execute_reply.started":"2023-05-01T13:33:43.447432Z","shell.execute_reply":"2023-05-01T13:33:43.452294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sanity Check\nif TRAIN_MODEL and USE_VAL:\n    _ = model.evaluate(*validation_data, verbose=2)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:43.455664Z","iopub.execute_input":"2023-05-01T13:33:43.456655Z","iopub.status.idle":"2023-05-01T13:33:43.466457Z","shell.execute_reply.started":"2023-05-01T13:33:43.456616Z","shell.execute_reply":"2023-05-01T13:33:43.465268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\nmodel1 = get_model()\n# Plot model summary\nmodel1.summary(expand_nested=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:43.469612Z","iopub.execute_input":"2023-05-01T13:33:43.470761Z","iopub.status.idle":"2023-05-01T13:33:45.399324Z","shell.execute_reply.started":"2023-05-01T13:33:43.470718Z","shell.execute_reply":"2023-05-01T13:33:45.398521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    # Clear all models in GPU\n    tf.keras.backend.clear_session()\n\n    # Get new fresh model\n    model = get_model()\n    \n    # Sanity Check\n    model.summary()\n\n    # Actual Training\n    history = model.fit(\n            x=get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN),\n            steps_per_epoch=len(X_train) // (NUM_CLASSES * BATCH_ALL_SIGNS_N),\n            epochs=N_EPOCHS,\n            # Only used for validation data since training data is a generator\n            #\"只用于验证数据，因为训练数据是生成器\"。\n            #如果使用生成器来训练模型，则必须使用验证数据来评估模型的性能。\n            #这是因为生成器在每个时期中都会生成新的数据，而不是将所有数据加载到内存中。\n            #因此，无法在训练期间使用训练数据来评估模型的性能。相反，必须使用验证数据来评估模型的性能.\n            batch_size=BATCH_SIZE,\n            validation_data=validation_data,\n            callbacks=[\n                lr_callback,\n                WeightDecayCallback(),\n            ],\n            verbose = 2,\n        )","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:33:45.400758Z","iopub.execute_input":"2023-05-01T13:33:45.401315Z","iopub.status.idle":"2023-05-01T13:35:22.093464Z","shell.execute_reply.started":"2023-05-01T13:33:45.401277Z","shell.execute_reply":"2023-05-01T13:35:22.092288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save Model Weights\nmodel.save_weights('model.h5')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:22.098770Z","iopub.execute_input":"2023-05-01T13:35:22.101889Z","iopub.status.idle":"2023-05-01T13:35:22.344669Z","shell.execute_reply.started":"2023-05-01T13:35:22.101844Z","shell.execute_reply":"2023-05-01T13:35:22.343497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if USE_VAL:\n    # Validation Predictions\n    y_val_pred = model.predict({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, verbose=2).argmax(axis=1)\n    # Label\n    labels = [ORD2SIGN.get(i).replace(' ', '_') for i in range(NUM_CLASSES)]","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:22.349695Z","iopub.execute_input":"2023-05-01T13:35:22.350706Z","iopub.status.idle":"2023-05-01T13:35:22.358735Z","shell.execute_reply.started":"2023-05-01T13:35:22.350663Z","shell.execute_reply":"2023-05-01T13:35:22.357737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Landmark Weights\nfor w in model.get_layer('embedding').weights:\n    if 'landmark_weights' in w.name:\n        weights = scipy.special.softmax(w)\n\nlandmarks = ['lips_embedding', 'left_hand_embedding', 'pose_embedding']\n\nfor w, lm in zip(weights, landmarks):\n    print(f'{lm} weight: {(w*100):.1f}%')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:22.363012Z","iopub.execute_input":"2023-05-01T13:35:22.365935Z","iopub.status.idle":"2023-05-01T13:35:22.378965Z","shell.execute_reply.started":"2023-05-01T13:35:22.365881Z","shell.execute_reply":"2023-05-01T13:35:22.377968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_classification_report():\n    # Classification report for all signs\n    classification_report = sklearn.metrics.classification_report(\n            y_val,\n            y_val_pred,\n            target_names=labels,\n            output_dict=True,\n        )\n    # Round Data for better readability\n    classification_report = pd.DataFrame(classification_report).T\n    classification_report = classification_report.round(2)\n    classification_report = classification_report.astype({\n            'support': np.uint16,\n        })\n    # Add signs\n    classification_report['sign'] = [e if e in SIGN2ORD else -1 for e in classification_report.index]\n    classification_report['sign_ord'] = classification_report['sign'].apply(SIGN2ORD.get).fillna(-1).astype(np.int16)\n    # Sort on F1-score\n    classification_report = pd.concat((\n        classification_report.head(NUM_CLASSES).sort_values('f1-score', ascending=False),\n        classification_report.tail(3),\n    ))\n\n    pd.options.display.max_rows = 999\n    display(classification_report)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:22.383263Z","iopub.execute_input":"2023-05-01T13:35:22.386133Z","iopub.status.idle":"2023-05-01T13:35:22.397591Z","shell.execute_reply.started":"2023-05-01T13:35:22.386098Z","shell.execute_reply":"2023-05-01T13:35:22.396566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if USE_VAL:\n    print_classification_report()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:22.401975Z","iopub.execute_input":"2023-05-01T13:35:22.404876Z","iopub.status.idle":"2023-05-01T13:35:22.412128Z","shell.execute_reply.started":"2023-05-01T13:35:22.404829Z","shell.execute_reply":"2023-05-01T13:35:22.411166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_history_metric(metric, f_best=np.argmax, ylim=None, yscale=None, yticks=None):\n    plt.figure(figsize=(20, 10))\n    \n    values = history.history[metric]\n    N_EPOCHS = len(values)\n    val = 'val' in ''.join(history.history.keys())\n    # Epoch Ticks\n    if N_EPOCHS <= 20:\n        x = np.arange(1, N_EPOCHS + 1)\n    else:\n        x = [1, 5] + [10 + 5 * idx for idx in range((N_EPOCHS - 10) // 5 + 1)]\n\n    x_ticks = np.arange(1, N_EPOCHS+1)\n\n    # Validation\n    if val:\n        val_values = history.history[f'val_{metric}']\n        val_argmin = f_best(val_values)\n        plt.plot(x_ticks, val_values, label=f'val')\n\n    # summarize history for accuracy\n    plt.plot(x_ticks, values, label=f'train')\n    argmin = f_best(values)\n    plt.scatter(argmin + 1, values[argmin], color='red', s=75, marker='o', label=f'train_best')\n    if val:\n        plt.scatter(val_argmin + 1, val_values[val_argmin], color='purple', s=75, marker='o', label=f'val_best')\n\n    plt.title(f'Model {metric}', fontsize=24, pad=10)\n    plt.ylabel(metric, fontsize=20, labelpad=10)\n\n    if ylim:\n        plt.ylim(ylim)\n\n    if yscale is not None:\n        plt.yscale(yscale)\n        \n    if yticks is not None:\n        plt.yticks(yticks, fontsize=16)\n\n    plt.xlabel('epoch', fontsize=20, labelpad=10)        \n    plt.tick_params(axis='x', labelsize=8)\n    plt.xticks(x, fontsize=16) # set tick step to 1 and let x axis start at 1\n    plt.yticks(fontsize=16)\n    \n    plt.legend(prop={'size': 10})\n    plt.grid()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:22.416757Z","iopub.execute_input":"2023-05-01T13:35:22.419901Z","iopub.status.idle":"2023-05-01T13:35:22.436421Z","shell.execute_reply.started":"2023-05-01T13:35:22.419853Z","shell.execute_reply":"2023-05-01T13:35:22.435401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('loss', f_best=np.argmin)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:22.440830Z","iopub.execute_input":"2023-05-01T13:35:22.443425Z","iopub.status.idle":"2023-05-01T13:35:22.810832Z","shell.execute_reply.started":"2023-05-01T13:35:22.443389Z","shell.execute_reply":"2023-05-01T13:35:22.809763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1))","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:22.812674Z","iopub.execute_input":"2023-05-01T13:35:22.813111Z","iopub.status.idle":"2023-05-01T13:35:23.106457Z","shell.execute_reply.started":"2023-05-01T13:35:22.813071Z","shell.execute_reply":"2023-05-01T13:35:23.104981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('top_5_acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1))","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:23.108783Z","iopub.execute_input":"2023-05-01T13:35:23.110220Z","iopub.status.idle":"2023-05-01T13:35:23.410988Z","shell.execute_reply.started":"2023-05-01T13:35:23.110156Z","shell.execute_reply":"2023-05-01T13:35:23.409887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    plot_history_metric('top_10_acc', ylim=[0,1], yticks=np.arange(0.0, 1.1, 0.1))","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:23.412586Z","iopub.execute_input":"2023-05-01T13:35:23.414789Z","iopub.status.idle":"2023-05-01T13:35:23.718869Z","shell.execute_reply.started":"2023-05-01T13:35:23.414739Z","shell.execute_reply":"2023-05-01T13:35:23.717848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # TFLite model for submission\n# class TFLiteModel(tf.Module):\n#     def __init__(self, model):\n#         super(TFLiteModel, self).__init__()\n\n#         # Load the feature generation and main models\n#         self.preprocess_layer = preprocess_layer\n#         self.model = model\n    \n#     @tf.function(input_signature=[tf.TensorSpec(shape=[None, N_ROWS, N_DIMS], dtype=tf.float32, name='inputs')])\n#     def __call__(self, inputs):\n#         # Preprocess Data\n#         x, non_empty_frame_idxs = self.preprocess_layer(inputs)\n#         # Add Batch Dimension\n#         x = tf.expand_dims(x, axis=0)\n#         non_empty_frame_idxs = tf.expand_dims(non_empty_frame_idxs, axis=0)\n#         # Make Prediction\n#         outputs = self.model({ 'frames': x, 'non_empty_frame_idxs': non_empty_frame_idxs })\n#         # Squeeze Output 1x250 -> 250\n#         outputs = tf.squeeze(outputs, axis=0)\n\n#         # Return a dictionary with the output tensor\n#         return {'outputs': outputs}\n\n# # Define TF Lite Model\n# tflite_keras_model = TFLiteModel(model)\n\n# # Sanity Check\n# demo_raw_data = load_relevant_data_subset(train['file_path'].values[5])\n# print(f'demo_raw_data shape: {demo_raw_data.shape}, dtype: {demo_raw_data.dtype}')\n# demo_output = tflite_keras_model(demo_raw_data)[\"outputs\"]\n# print(f'demo_output shape: {demo_output.shape}, dtype: {demo_output.dtype}')\n# demo_prediction = demo_output.numpy().argmax()\n# print(f'demo_prediction: {demo_prediction}, correct: {train.iloc[0][\"sign_ord\"]}')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:23.720729Z","iopub.execute_input":"2023-05-01T13:35:23.721418Z","iopub.status.idle":"2023-05-01T13:35:23.727385Z","shell.execute_reply.started":"2023-05-01T13:35:23.721375Z","shell.execute_reply":"2023-05-01T13:35:23.726269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Create Model Converter\n# keras_model_converter = tf.lite.TFLiteConverter.from_keras_model(tflite_keras_model)\n# # Convert Model\n# tflite_model = keras_model_converter.convert()\n# # Write Model\n# with open('/kaggle/working/model.tflite', 'wb') as f:\n#     f.write(tflite_model)\n    \n# # Zip Model\n# !zip submission.zip /kaggle/working/model.tflite","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:23.728868Z","iopub.execute_input":"2023-05-01T13:35:23.729538Z","iopub.status.idle":"2023-05-01T13:35:23.741028Z","shell.execute_reply.started":"2023-05-01T13:35:23.729485Z","shell.execute_reply":"2023-05-01T13:35:23.739643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(FeatureGen()(tf.keras.Input((543, 3), dtype=tf.float32, name=\"inputs\")))\n# FeatureGen()(load_relevant_data_subset(train.path[0]))","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:23.742797Z","iopub.execute_input":"2023-05-01T13:35:23.743279Z","iopub.status.idle":"2023-05-01T13:35:23.752226Z","shell.execute_reply.started":"2023-05-01T13:35:23.743240Z","shell.execute_reply":"2023-05-01T13:35:23.751242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Class Count\n# display(pd.Series(y_train0).value_counts().to_frame('Class Count').iloc[[0,1,2,3,4, -5,-4,-3,-2,-1]])","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:23.753910Z","iopub.execute_input":"2023-05-01T13:35:23.754441Z","iopub.status.idle":"2023-05-01T13:35:23.762040Z","shell.execute_reply.started":"2023-05-01T13:35:23.754385Z","shell.execute_reply":"2023-05-01T13:35:23.761311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# feature_gen=FeatureGen()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:23.763460Z","iopub.execute_input":"2023-05-01T13:35:23.764147Z","iopub.status.idle":"2023-05-01T13:35:23.775333Z","shell.execute_reply.started":"2023-05-01T13:35:23.764105Z","shell.execute_reply":"2023-05-01T13:35:23.774041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LIPS_START = 0\n# LEFT_HAND_START = LIPS_IDXS.size\n# RIGHT_HAND_START = LEFT_HAND_START + LEFT_HAND_IDXS.size\n# POSE_START = RIGHT_HAND_START + RIGHT_HAND_IDXS.size\n\n# def get_data_features(file_path):\n#     data = load_relevant_data_subset(file_path)\n#     data = feature_gen(data)\n#     return data\n# def get_x_y_features():\n#     # Create arrays to save data\n#     X = np.zeros([len(train), FLAT_INPUT_SHAPE], dtype=np.float32)\n#     y = np.zeros([len(train)], dtype=np.int32)\n\n#     for row_idx, (file_path, sign_ord) in enumerate(tqdm(train[['file_path', 'sign_ord']].values)):\n#         if row_idx % 5000 == 0:\n#             print(f'Generated {row_idx}/{N_SAMPLES}')\n#         data = get_data_features(file_path)\n#         X[row_idx] = data\n#         y[row_idx] = sign_ord\n#         if np.isnan(data).sum() > 0: return data\n#         # Save X/y\n#     np.save('X.npy', X)\n#     np.save('y.npy', y)\n#     return X, y","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:23.777066Z","iopub.execute_input":"2023-05-01T13:35:23.777545Z","iopub.status.idle":"2023-05-01T13:35:23.787125Z","shell.execute_reply.started":"2023-05-01T13:35:23.777473Z","shell.execute_reply":"2023-05-01T13:35:23.786035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(model1, show_shapes=True, show_dtype=True, show_layer_names=True, expand_nested=True, show_layer_activations=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:23.790600Z","iopub.execute_input":"2023-05-01T13:35:23.791170Z","iopub.status.idle":"2023-05-01T13:35:24.640840Z","shell.execute_reply.started":"2023-05-01T13:35:23.791141Z","shell.execute_reply":"2023-05-01T13:35:24.639413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#del X_train0; del y_train0","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:24.643156Z","iopub.execute_input":"2023-05-01T13:35:24.643590Z","iopub.status.idle":"2023-05-01T13:35:24.648973Z","shell.execute_reply.started":"2023-05-01T13:35:24.643548Z","shell.execute_reply":"2023-05-01T13:35:24.647663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#del X_val; del y_val; del NON_EMPTY_FRAME_IDXS_VAL; del NON_EMPTY_FRAME_IDXS_TRAIN0","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:24.651142Z","iopub.execute_input":"2023-05-01T13:35:24.651488Z","iopub.status.idle":"2023-05-01T13:35:24.660469Z","shell.execute_reply.started":"2023-05-01T13:35:24.651455Z","shell.execute_reply":"2023-05-01T13:35:24.659093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train0 = np.load(f'{ROOT_DIR}/X_train.npy')\n# y_train0 = np.load(f'{ROOT_DIR}/y_train.npy')\n# NON_EMPTY_FRAME_IDXS_TRAIN0 = np.load(f'{ROOT_DIR}/NON_EMPTY_FRAME_IDXS_TRAIN.npy')\n# validation_data = None","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:24.662520Z","iopub.execute_input":"2023-05-01T13:35:24.662978Z","iopub.status.idle":"2023-05-01T13:35:24.671198Z","shell.execute_reply.started":"2023-05-01T13:35:24.662939Z","shell.execute_reply":"2023-05-01T13:35:24.669984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TRAIN_MODEL:\n    # Clear all models in GPU\n    tf.keras.backend.clear_session()\n    \n\n    # Actual Training\n    history = model1.fit(\n            x=get_train_batch_all_signs(X_train, y_train, NON_EMPTY_FRAME_IDXS_TRAIN),\n            steps_per_epoch=len(X_train) // (NUM_CLASSES * BATCH_ALL_SIGNS_N),\n            epochs=N_EPOCHS,\n            # Only used for validation data since training data is a generator\n            #\"只用于验证数据，因为训练数据是生成器\"。\n            #如果使用生成器来训练模型，则必须使用验证数据来评估模型的性能。\n            #这是因为生成器在每个时期中都会生成新的数据，而不是将所有数据加载到内存中。\n            #因此，无法在训练期间使用训练数据来评估模型的性能。相反，必须使用验证数据来评估模型的性能.\n            batch_size=BATCH_SIZE,\n            validation_data=validation_data,\n            callbacks=[\n                lr_callback,\n                WeightDecayCallback(),\n            ],\n            verbose = 2,\n        )","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:35:24.673375Z","iopub.execute_input":"2023-05-01T13:35:24.673806Z","iopub.status.idle":"2023-05-01T13:36:59.783986Z","shell.execute_reply.started":"2023-05-01T13:35:24.673770Z","shell.execute_reply":"2023-05-01T13:36:59.782409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X_train; del y_train; del NON_EMPTY_FRAME_IDXS_TRAIN","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:36:59.786394Z","iopub.execute_input":"2023-05-01T13:36:59.787447Z","iopub.status.idle":"2023-05-01T13:36:59.797480Z","shell.execute_reply.started":"2023-05-01T13:36:59.787396Z","shell.execute_reply":"2023-05-01T13:36:59.796011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocess_layer=PreprocessLayer()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# inference_model = get_inference_model(ltsm_model,asl_model)\n# tf.keras.utils.plot_model(inference_model)\nclass FinalModel(tf.Module):\n    def __init__(self, model_1, model_2):\n        super().__init__()\n        self.model_1 = model_1\n        #self.model_2= model_2\n        self.model_3= model_2\n        self.pp_layer_1 = preprocess_layer\n        #self.pp_layer_2 = PreprocessLayer()\n        self.pp_layer_3 = preprocess_layer\n    @tf.function(input_signature=[tf.TensorSpec(shape=[None, N_ROWS, N_DIMS], dtype=tf.float32, name='inputs')])        \n    def __call__(self, inputs):\n        #model-2 (transformer)\n        x, non_empty_frame_idxs = self.pp_layer_1(inputs)\n        x = tf.expand_dims(x, axis=0)\n        non_empty_frame_idxs = tf.expand_dims(non_empty_frame_idxs, axis=0)\n      \n       #Transfomer1  \n        _outputs_1 = self.model_1({'frames':x, 'non_empty_frame_idxs':non_empty_frame_idxs})\n        _outputs_1 = tf.squeeze(_outputs_1, axis=0)\n        \n        \n#         #Transformer2\n#         x, non_empty_frame_idxs = self.pp_layer_3(inputs)\n#         x = tf.expand_dims(x, axis=0)\n#         non_empty_frame_idxs = tf.expand_dims(non_empty_frame_idxs, axis=0)\n        _outputs_2 = self.model_3({'frames':x, 'non_empty_frame_idxs':non_empty_frame_idxs})\n        _outputs_2 = tf.squeeze(_outputs_2, axis=0)\n        \n        outputs = 0.55*_outputs_1 + 0.45 * _outputs_2\n       # print( outputs.shape,  outputs)\n        # Return a dictionary with the output tensor\n        return {'outputs': outputs}","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:36:59.799590Z","iopub.execute_input":"2023-05-01T13:36:59.800558Z","iopub.status.idle":"2023-05-01T13:36:59.819576Z","shell.execute_reply.started":"2023-05-01T13:36:59.800515Z","shell.execute_reply":"2023-05-01T13:36:59.818229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model = FinalModel(model, model1)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:36:59.822491Z","iopub.execute_input":"2023-05-01T13:36:59.823548Z","iopub.status.idle":"2023-05-01T13:36:59.857059Z","shell.execute_reply.started":"2023-05-01T13:36:59.823497Z","shell.execute_reply":"2023-05-01T13:36:59.845492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Sanity Check\ndemo_raw_data = load_relevant_data_subset(train['file_path'].values[0])\nprint(f'demo_raw_data shape: {demo_raw_data.shape}, dtype: {demo_raw_data.dtype}')\ndemo_output = final_model(demo_raw_data)[\"outputs\"]\n#print(f'demo_output shape: {demo_output.shape}, dtype: {demo_output.dtype}')\ndemo_prediction = np.array(demo_output).argmax()\nprint(f'demo_prediction: {demo_prediction}, correct: {train.iloc[0][\"sign_ord\"]}')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:36:59.860389Z","iopub.execute_input":"2023-05-01T13:36:59.860728Z","iopub.status.idle":"2023-05-01T13:37:03.400752Z","shell.execute_reply.started":"2023-05-01T13:36:59.860689Z","shell.execute_reply":"2023-05-01T13:37:03.399706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# demo_output.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:37:03.411434Z","iopub.execute_input":"2023-05-01T13:37:03.411739Z","iopub.status.idle":"2023-05-01T13:37:03.418900Z","shell.execute_reply.started":"2023-05-01T13:37:03.411711Z","shell.execute_reply":"2023-05-01T13:37:03.417775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# import tensorflow as tf\n# import zipfile\n\n# import numpy as np\n# import tensorflow as tf\n# from tensorflow import lite\n# # Load the two TFLite models\n# model1_path = '/kaggle/input/transformer-model-training/model.tflite'\n# model2_path = '/kaggle/input/transformer-model-training-87404b/model.tflite'\n\n# interpreter1 = lite.Interpreter(model_path=model1_path)\n# interpreter2 = lite.Interpreter(model_path=model2_path)\n\n# # Allocate memory for the models\n# interpreter1.allocate_tensors()\n# interpreter2.allocate_tensors()\n\n# # Get the input and output tensors for both models\n# input_details1 = interpreter1.get_input_details()\n# output_details1 = interpreter1.get_output_details()\n\n# input_details2 = interpreter2.get_input_details()\n# output_details2 = interpreter2.get_output_details()\n\n\n# # Load your input data for both models here\n# input_data1 = np.random.rand(1, 543, 3).astype(np.float32)\n# input_data2 = np.random.rand(1, 543, 3).astype(np.float32)\n\n# # Get the input and output details for both models\n# input_details1 = interpreter1.get_input_details()\n# output_details1 = interpreter1.get_output_details()\n\n# input_details2 = interpreter2.get_input_details()\n# output_details2 = interpreter2.get_output_details()\n\n# # Run the first input data through the first model\n# interpreter1.set_tensor(input_details1[0]['index'], input_data1)\n# interpreter1.invoke()\n# output1 = interpreter1.get_tensor(output_details1[0]['index'])\n\n# # Run the second input data through the second model\n# interpreter2.set_tensor(input_details2[0]['index'], input_data2)\n# interpreter2.invoke()\n# output2 = interpreter2.get_tensor(output_details2[0]['index'])\n\n# # Concatenate the outputs along the batch dimension\n# ensemble_output = np.concatenate((output1, output2), axis=0)\n# converter = tf.lite.TFLiteConverter.from_keras_model(model)\n# converter.optimizations = [tf.lite.Optimize.DEFAULT]\n# tflite_model = converter.convert()\n\n# with open('ensemble_model.tflite', 'wb') as f:\n#     f.write(tflite_model)\n\n\n# zip_file_path = \"/kaggle/working/submission.zip\"\n\n# with open('ensemble_model.tflite', 'rb') as f:\n#     tflite_model = f.read()\n\n# with zipfile.ZipFile(zip_file_path, 'w', zipfile.ZIP_DEFLATED) as zip_file:\n#     zip_file.writestr(\"model.tflite\", tflite_model)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T13:37:03.422530Z","iopub.execute_input":"2023-05-01T13:37:03.422806Z","iopub.status.idle":"2023-05-01T13:37:03.432647Z","shell.execute_reply.started":"2023-05-01T13:37:03.422780Z","shell.execute_reply":"2023-05-01T13:37:03.431720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"converter = tf.lite.TFLiteConverter.from_keras_model(final_model)\nconverter.optimizations = [tf.lite.Optimize.DEFAULT]\ntflite_model = converter.convert()\nwith open('./model.tflite', 'wb') as f:\n    f.write(tflite_model)\n#Zip Model\n!zip submission.zip /kaggle/working/model.tflite","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Verify TFLite model can be loaded and used for prediction\n# !pip install tflite-runtime\n# import tflite_runtime.interpreter as tflite\n\n# interpreter = tflite.Interpreter(\"/kaggle/working/model.tflite\")\n# found_signatures = list(interpreter.get_signature_list().keys())\n# print(found_signatures)  # Print the available signatures\n# prediction_fn = interpreter.get_signature_runner(\"serving_default\")\n\n# # Create dummy data to play with shapes\n# input_data = np.random.rand(78, 543, 3).astype(np.float32)\n\n# output = prediction_fn(inputs=input_data)\n# sign = output['outputs'].argmax()\n\n# print(\"PRED : \", ORD2SIGN.get(sign), f'[{sign}]')\n# print(\"TRUE : \", train.sign.values[0], f'[{train.sign_ord.values[0]}]')","metadata":{},"execution_count":null,"outputs":[]}]}