{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nn = 0\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    if n > 100:\n        break\n        \n    for filename in filenames:\n        if n > 100:\n            break \n            \n        print(os.path.join(dirname, filename))\n        n += 1\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-30T14:48:29.990750Z","iopub.execute_input":"2023-04-30T14:48:29.991538Z","iopub.status.idle":"2023-04-30T14:48:30.368838Z","shell.execute_reply.started":"2023-04-30T14:48:29.991494Z","shell.execute_reply":"2023-04-30T14:48:30.367389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n#%matplotlib widget  # comment out for notebook visualization\nimport time\nfrom mpl_toolkits.mplot3d import Axes3D\nfrom matplotlib.animation import FuncAnimation\nimport plotly.express as px\npath_train = '/kaggle/input/asl-signs/train.csv'\ndftr = pd.read_csv(path_train)\n\n#-Takes a single landmark \nexample_landmark = pd.read_parquet('/kaggle/input/asl-signs/train_landmark_files/26734/1000035562.parquet')\nprint(example_landmark)\nminn = example_landmark['frame'].min()#-We need to start at first and end at last frame of where this landmark occurs\nmaxx = example_landmark['frame'].max()\n\n#-For every frame in the video...\nfor i in range(minn, maxx+1):\n    print('Frame: ' + str(i))\n    #-t is set to be only the landmark where the frame matches. This is still a dataframe.\n    t = example_landmark[example_landmark['frame'] == i] \n#     print(t)\n    #-Now we only print the sum of the x cordinates. WHY?\n    print('   Face: ' + str(t[t['type'] == 'face']['x'].isnull().sum()))\n    print('   Pose: ' + str(t[t['type'] == 'pose']['x'].isnull().sum()))\n    print('   RH: ' + str(t[t['type'] == 'right_hand']['x'].isnull().sum()))\n    print('   LH: ' + str(t[t['type'] == 'left_hand']['x'].isnull().sum()))","metadata":{"execution":{"iopub.status.busy":"2023-04-30T14:48:30.371729Z","iopub.execute_input":"2023-04-30T14:48:30.372737Z","iopub.status.idle":"2023-04-30T14:48:34.163698Z","shell.execute_reply.started":"2023-04-30T14:48:30.372699Z","shell.execute_reply":"2023-04-30T14:48:34.162278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sgl_in = example_landmark[example_landmark['frame'] == 21]\nsgl_in['type'].value_counts()\nsgl_in_rh = sgl_in[sgl_in['type'] == 'right_hand']\n\n#-\ndef add_init_c(start, end, hand):\n    return (\n        pd.concat([hand['x'][start:start+1], hand['x'][end[0]:end[1]]]), \n        pd.concat([hand['y'][start:start+1], hand['y'][end[0]:end[1]]]), \n        pd.concat([hand['z'][start:start+1], hand['z'][end[0]:end[1]]])\n        )\ndef plot_hand(hand, td):\n    fig, ax = plt.subplots()\n    fig.set_size_inches(6, 6)\n\n    if td:\n        ax = Axes3D(fig, auto_add_to_figure=False)\n        fig.add_axes(ax)\n\n    ind = [[0, [0, 5]], [0, [5, 9]], [0, [17, 21]], [5, [9, 13]], [17, [13, 17]], [9, [13, 14]]]\n\n    for i, k in ind: \n        x, y, z = add_init_c(i, k, hand)\n        if td:\n            ax.plot(x, -1*y, z)\n        else:\n            ax.plot(x, -1*y)\nplot_hand(sgl_in_rh, True)","metadata":{"execution":{"iopub.status.busy":"2023-04-30T14:48:34.164989Z","iopub.execute_input":"2023-04-30T14:48:34.165320Z","iopub.status.idle":"2023-04-30T14:48:34.730156Z","shell.execute_reply.started":"2023-04-30T14:48:34.165290Z","shell.execute_reply":"2023-04-30T14:48:34.729207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sgl_in_lh = sgl_in[sgl_in['type'] == 'left_hand']\nsgl_in_p = sgl_in[sgl_in['type'] == 'pose']\ndef plot_pose(pose, td):\n    fig, ax = plt.subplots()\n    fig.set_size_inches(6, 6)\n\n    if td:\n        ax = Axes3D(fig, auto_add_to_figure=False)\n        fig.add_axes(ax)\n\n    ind = [[0, [1, 4]], [3, [7, 8]], [0, [4, 7]], [6, [8, 9]], [9, [10, 11]], [11, [12, 13]], [12, [14, 15]], [14, [16, 17]], \n           [16, [22, 23]], [16, [18, 19]], [16, [20, 21]], [18, [20, 21]], [11, [13, 14]], [13, [15, 16]], [15, [21, 22]], [15, [19, 20]],\n           [15, [17, 18]], [17, [19, 20]], [12, [24, 25]], [24, [26, 27]], [26, [28, 29]], [28, [30, 31]], [30, [32, 33]], [28, [32, 33]], \n           [11, [23, 24]], [23, [25, 26]], [25, [27, 28]], [27, [29, 30]], [29, [31, 32]], [27, [31, 32]], [23, [24, 25]]]\n\n    for i, k in ind: \n        x, y, z = add_init_c(i, k, pose)\n        if td:\n            ax.plot(x, -1*y, z)\n        else:\n            ax.plot(x, -1*y)\nplot_pose(sgl_in_p, True)\nplot_pose(sgl_in_p, False)\nsgl_in_f = sgl_in[sgl_in['type'] == 'face']\ndef plot_face(face, td):\n    fig, ax = plt.subplots()\n    fig.set_size_inches(6, 6)\n\n    if td:\n        ax = Axes3D(fig, auto_add_to_figure=False)\n        fig.add_axes(ax)\n\n    if td:\n        ax.scatter(face['x'], -1*face['y'], face['z'])\n    else:\n        ax.scatter(face['x'], -1*face['y'])\nplot_face(sgl_in_f, True)\ndef plot_mix(pose, face, rhand, lhand, td):\n    fig, ax = plt.subplots()\n    fig.set_size_inches(6, 6)\n\n    ind = [[11, [12, 13]], [12, [14, 15]], [14, [16, 17]], [11, [13, 14]], [13, [15, 16]], [12, [24, 25]], \n           [24, [26, 27]], [26, [28, 29]], [28, [30, 31]], [30, [32, 33]], [28, [32, 33]], [11, [23, 24]], [23, [25, 26]], [25, [27, 28]], \n           [27, [29, 30]], [29, [31, 32]], [27, [31, 32]], [23, [24, 25]]]\n\n    if td:\n        ax = Axes3D(fig, auto_add_to_figure=False)\n        fig.add_axes(ax)\n\n    for i, k in ind: \n        x, y, z = add_init_c(i, k, pose)\n        if td:\n            ax.plot(x, -1*y, z)\n        else:\n            ax.plot(x, -1*y)\n            \n    s = [1] * face['x'].shape[0]\n    if td:\n        ax.scatter(face['x'], -1*face['y'], face['z'], s=s)\n    else:\n        ax.scatter(face['x'], -1*face['y'], s=s)\n        \n    ind = [[0, [0, 5]], [0, [5, 9]], [0, [17, 21]], [5, [9, 13]], [17, [13, 17]], [9, [13, 14]]]\n    \n    for i, k in ind: \n        x, y, z = add_init_c(i, k, rhand)\n        if td:\n            ax.plot(x, -1*y, z)\n        else:\n            ax.plot(x, -1*y)\n    \n    for i, k in ind: \n        x, y, z = add_init_c(i, k, lhand)\n        if td:\n            ax.plot(x, -1*y, z)\n        else:\n            ax.plot(x, -1*y)\nplot_mix(sgl_in_p, sgl_in_f, sgl_in_rh, sgl_in_lh, True)","metadata":{"execution":{"iopub.status.busy":"2023-04-30T14:48:34.731993Z","iopub.execute_input":"2023-04-30T14:48:34.732344Z","iopub.status.idle":"2023-04-30T14:48:36.855645Z","shell.execute_reply.started":"2023-04-30T14:48:34.732309Z","shell.execute_reply":"2023-04-30T14:48:36.854258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"face_an = example_landmark[example_landmark['type'] == 'face']\nfig = px.scatter_3d(face_an, x=\"x\", y=\"y\", z=\"z\", animation_frame=\"frame\", hover_name=\"row_id\", width=800, height=800)\nfig.update_traces(marker_size=3)\n\nfig.update_layout(\n    scene = dict(\n        xaxis = dict(visible=False),\n        yaxis = dict(visible=False),\n        zaxis =dict(visible=False)\n        )\n    )\n\ncamera_params = dict(\n    up=dict(x=0,y=0,z=0),\n    center=dict(x=0,y=0,z=0),\n    eye=dict(x=0,y=0,z=-2)\n)\n\nfig.update_layout(scene_camera=camera_params)\nface_an['y'] = face_an['y'] * -1\n\nfig = px.scatter(face_an, x=\"x\", y=\"y\", animation_frame=\"frame\", hover_name=\"row_id\", width=800, height=800)\nfig.update_traces(marker_size=3)\n\nfig.update_layout(\n    scene = dict(\n        xaxis = dict(visible=False),\n        yaxis = dict(visible=False),\n        )\n    )\n\ncamera_params = dict(\n    up=dict(x=0,y=0),\n    center=dict(x=0,y=0),\n    eye=dict(x=0,y=0)\n)\n\nfig.update_layout(scene_camera=camera_params)\npose_place_ind = [0, 1, 1, 1, 2, 2, 2, 1, 2, 3, 3, 4, 5, 4, 5, 4, 5, 4, 5, 4, 5, 6, 7, 8, 9, 8, 9, 8, 9, 8, 9, 8, 9] * 23\npose_an = example_landmark[example_landmark['type'] == 'pose']\npose_an['pose_place'] = pose_place_ind\nfig = px.line_3d(pose_an, x=\"x\", y=\"y\", z=\"z\", animation_frame=\"frame\", hover_name=\"row_id\", color=\"pose_place\", width=800, height=800)\nfig.update_traces(marker_size=3)\n\nfig.update_layout(\n    scene = dict(\n        xaxis = dict(visible=False),\n        yaxis = dict(visible=False),\n        zaxis =dict(visible=False)\n        )\n    )\n\ncamera_params = dict(\n    up=dict(x=0,y=0,z=0),\n    center=dict(x=0,y=0,z=0),\n    eye=dict(x=0,y=0,z=-2)\n)\n\nfig.update_layout(scene_camera=camera_params)\npose_an['y'] = pose_an['y'] * -1\n\nfig = px.line(pose_an, x=\"x\", y=\"y\", animation_frame=\"frame\", hover_name=\"row_id\", color=\"pose_place\", width=800, height=800)\nfig.update_traces(marker_size=3)\n\nfig.update_layout(\n    scene = dict(\n        xaxis = dict(visible=False),\n        yaxis = dict(visible=False),\n        )\n    )\n\ncamera_params = dict(\n    up=dict(x=0,y=0),\n    center=dict(x=0,y=0),\n    eye=dict(x=0,y=0)\n)\n\nfig.update_layout(scene_camera=camera_params)\nhand_place = []\n\ndef create_hand(row):\n    lnd = row['landmark_index']\n    if lnd < 5:\n        hand_place.append(1)\n    elif 5 <= lnd < 9:\n        hand_place.append(2)\n    elif 9 <= lnd < 13:\n        hand_place.append(3)\n    elif 13 <= lnd < 17:\n        hand_place.append(4)\n    else:\n        hand_place.append(5)\nrh_an = example_landmark[example_landmark['type'] == 'right_hand']\nrh_an.apply(create_hand, axis=1)\nrh_an['hand_place'] = hand_place\nfig = px.line_3d(rh_an, x=\"x\", y=\"y\", z='z', animation_frame=\"frame\", hover_name=\"row_id\", color=\"hand_place\", width=800, height=800)\nfig.update_traces(marker_size=3)\n\nfig.update_layout(\n    scene = dict(\n        xaxis = dict(visible=False),\n        yaxis = dict(visible=False),\n        zaxis =dict(visible=False)\n        )\n    )\n\ncamera_params = dict(\n    up=dict(x=0,y=0,z=0),\n    center=dict(x=0,y=0,z=0),\n    eye=dict(x=0,y=0,z=-2)\n)\n\nfig.update_layout(scene_camera=camera_params)\nrh_an['y'] = rh_an['y'] * -1\n\nfig = px.line(rh_an, x=\"x\", y=\"y\", animation_frame=\"frame\", hover_name=\"row_id\", color=\"hand_place\", width=800, height=800)\nfig.update_traces(marker_size=3)\n\nfig.update_layout(\n    scene = dict(\n        xaxis = dict(visible=False),\n        yaxis = dict(visible=False),\n        )\n    )\n\ncamera_params = dict(\n    up=dict(x=0,y=0),\n    center=dict(x=0,y=0),\n    eye=dict(x=0,y=0)\n)\n\nfig.update_layout(scene_camera=camera_params)","metadata":{"execution":{"iopub.status.busy":"2023-04-30T14:48:36.858933Z","iopub.execute_input":"2023-04-30T14:48:36.859477Z","iopub.status.idle":"2023-04-30T14:48:42.452604Z","shell.execute_reply.started":"2023-04-30T14:48:36.859434Z","shell.execute_reply":"2023-04-30T14:48:42.451473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nimport json\nfrom tqdm import tqdm\nimport numpy as np\nimport pandas as pd\n\nimport torch\nimport torch.nn as nn\nimport multiprocessing as mp","metadata":{"execution":{"iopub.status.busy":"2023-04-30T14:48:42.454676Z","iopub.execute_input":"2023-04-30T14:48:42.455384Z","iopub.status.idle":"2023-04-30T14:48:45.095465Z","shell.execute_reply.started":"2023-04-30T14:48:42.455343Z","shell.execute_reply":"2023-04-30T14:48:45.094510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LANDMARK_FILES_DIR = \"/kaggle/input/asl-signs/train_landmark_files\"\nTRAIN_FILE = \"/kaggle/input/asl-signs/train.csv\"\nlabel_map = json.load(open(\"/kaggle/input/asl-signs/sign_to_prediction_index_map.json\", \"r\"))\nclass FeatureGen(nn.Module):\n    def __init__(self):\n        super(FeatureGen, self).__init__()\n        pass\n    \n    def forward(self, x):\n        face_x = x[:,:468,:].contiguous().view(-1, 468*3)\n        lefth_x = x[:,468:489,:].contiguous().view(-1, 21*3)\n        pose_x = x[:,489:522,:].contiguous().view(-1, 33*3)\n        righth_x = x[:,522:,:].contiguous().view(-1, 21*3)\n        \n        lefth_x = lefth_x[~torch.any(torch.isnan(lefth_x), dim=1),:]\n        righth_x = righth_x[~torch.any(torch.isnan(righth_x), dim=1),:]\n        \n        x1m = torch.mean(face_x, 0)\n        x2m = torch.mean(lefth_x, 0)\n        x3m = torch.mean(pose_x, 0)\n        x4m = torch.mean(righth_x, 0)\n        \n        x1s = torch.std(face_x, 0)\n        x2s = torch.std(lefth_x, 0)\n        x3s = torch.std(pose_x, 0)\n        x4s = torch.std(righth_x, 0)\n        \n        xfeat = torch.cat([x1m,x2m,x3m,x4m, x1s,x2s,x3s,x4s], axis=0)\n        xfeat = torch.where(torch.isnan(xfeat), torch.tensor(0.0, dtype=torch.float32), xfeat)\n        \n        return xfeat\n    \nfeature_converter = FeatureGen()","metadata":{"execution":{"iopub.status.busy":"2023-04-30T14:48:45.097032Z","iopub.execute_input":"2023-04-30T14:48:45.098106Z","iopub.status.idle":"2023-04-30T14:48:45.115050Z","shell.execute_reply.started":"2023-04-30T14:48:45.098064Z","shell.execute_reply":"2023-04-30T14:48:45.113896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROWS_PER_FRAME = 543\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)\ndef convert_row(row):\n    x = load_relevant_data_subset(os.path.join(\"/kaggle/input/asl-signs\", row[1].path))\n    x = feature_converter(torch.tensor(x)).cpu().numpy()\n    return x, row[1].label\n\ndef convert_and_save_data():\n    df = pd.read_csv(TRAIN_FILE)\n    df['label'] = df['sign'].map(label_map)\n    npdata = np.zeros((df.shape[0], 3258))\n    nplabels = np.zeros(df.shape[0])\n    with mp.Pool() as pool:\n        results = pool.imap(convert_row, df.iterrows(), chunksize=250)\n        for i, (x,y) in tqdm(enumerate(results), total=df.shape[0]):\n            npdata[i,:] = x\n            nplabels[i] = y\n    \n    np.save(\"feature_data.npy\", npdata)\n    np.save(\"feature_labels.npy\", nplabels)\n        \nconvert_and_save_data()","metadata":{"execution":{"iopub.status.busy":"2023-04-30T14:48:45.116470Z","iopub.execute_input":"2023-04-30T14:48:45.116939Z","iopub.status.idle":"2023-04-30T14:58:32.207779Z","shell.execute_reply.started":"2023-04-30T14:48:45.116901Z","shell.execute_reply":"2023-04-30T14:58:32.206230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q --upgrade tensorflow-io\n!pip install tflite-runtime\n!pip install ipdb\n!pip install wandb\n!pip install hydra-core\n!mkdir models\n!cp -r /kaggle/input/gislr-extended-train-dataframe/extended_train.csv ./\n!cp -r /kaggle/input/asl-signs/train_landmark_files/16069/1004211348.parquet .","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:34:42.078867Z","iopub.execute_input":"2023-04-30T15:34:42.079660Z","iopub.status.idle":"2023-04-30T15:45:27.109620Z","shell.execute_reply.started":"2023-04-30T15:34:42.079594Z","shell.execute_reply":"2023-04-30T15:45:27.107523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile common_func.py\n\nimport json\nimport os\nimport warnings\n\nfrom sklearn import metrics\n\nwarnings.filterwarnings(\"ignore\")\nos.environ[\"TF_DETERMINISTIC_OPS\"] = \"1\"\nos.environ[\"TF_CUDNN_DETERMINISTIC\"] = \"1\"\nos.environ[\"TF_CPP_MIN_LOG_LEVEL\"] = \"2\"\nimport random\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nfrom sklearn.model_selection import StratifiedGroupKFold\nfrom tqdm.notebook import tqdm\nfrom wandb.keras import WandbCallback, WandbMetricsLogger\n\nSAVE_DIR = \"./models/\"\n# SAVE_DIR = \"/kaggle/working/models/\"\nif Path(\"/kaggle/input/asl-signs/\").exists():\n    DATA_DIR = \"/kaggle/input/asl-signs/\"\n    ROOT_PATH = \"/kaggle/input/islr-external-data/\"\n    CSV_PATH = \"./\"\nelse:\n    DATA_DIR = \"/scratch/smart_data/islr_data/\"\n    ROOT_PATH = \"/scratch/smart_data/\"\n    CSV_PATH = \"/scratch/smart_data/islr_data/\"\n\nPARQ_PATH = \"/kaggle/input/asl-signs/\"\nNUMPY_PATH = \"/scratch/smart_data/numpy_files/\"\nLANDMARK_FILES_DIR = f\"{ROOT_PATH}train_landmark_files\"\nTRAIN_FILE = f\"{CSV_PATH}extended_new.csv\"\n\nROWS_PER_FRAME =543\n\n# Data Generation ###################################################################################\n\ndef tf_nan_mean(x, axis=0):\n    return tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), x), axis=axis) / tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), tf.ones_like(x)), axis=axis)\n\ndef tf_nan_std(x, axis=0):\n    d = x - tf_nan_mean(x, axis=axis)\n    return tf.math.sqrt(tf_nan_mean(d * d, axis=axis))\n\ndef flatten_means_and_stds(x, axis=0):\n    # Get means and stds\n    x_mean = tf_nan_mean(x, axis=0)\n    x_std  = tf_nan_std(x,  axis=0)\n\n    x_out = tf.concat([x_mean, x_std], axis=0)\n    x_out = tf.reshape(x_out, (1, INPUT_SHAPE[1]*2))\n    x_out = tf.where(tf.math.is_finite(x_out), x_out, tf.zeros_like(x_out))\n    return x_out\n\nclass FeatureGen(tf.keras.layers.Layer):\n    def __init__(self):\n        super(FeatureGen, self).__init__()\n    \n    def call(self, x_in):\n#         print(right_hand_percentage(x))\n#         x_list = [tf.expand_dims(tf_nan_mean(x_in[:, av_set[0]:av_set[0]+av_set[1], :], axis=1), axis=1) for av_set in averaging_sets]\n#         x_list.append(tf.gather(x_in, point_landmarks, axis=1))\n#         x = tf.concat(x_list, 1)\n        x = tf.gather(x_in, point_landmarks, axis=1)\n\n        x_padded = x\n        for i in range(SEGMENTS):\n            p0 = tf.where( ((tf.shape(x_padded)[0] % SEGMENTS) > 0) & ((i % 2) != 0) , 1, 0)\n            p1 = tf.where( ((tf.shape(x_padded)[0] % SEGMENTS) > 0) & ((i % 2) == 0) , 1, 0)\n            paddings = [[p0, p1], [0, 0], [0, 0]]\n            x_padded = tf.pad(x_padded, paddings, mode=\"SYMMETRIC\")\n        x_list = tf.split(x_padded, SEGMENTS)\n        x_list = [flatten_means_and_stds(_x, axis=0) for _x in x_list]\n\n        x_list.append(flatten_means_and_stds(x, axis=0))\n        \n        ## Resize only dimension 0. Resize can't handle nan, so replace nan with that dimension's avg value to reduce impact.\n        x = tf.image.resize(tf.where(tf.math.is_finite(x), x, tf_nan_mean(x, axis=0)), [NUM_FRAMES, LANDMARKS])\n        x = tf.reshape(x, (1, INPUT_SHAPE[0]*INPUT_SHAPE[1]))\n        x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n        x_list.append(x)\n        x = tf.concat(x_list, axis=1)\n        return x\n\ndef convert_row(row, right_handed=True):\n    x = load_relevant_data_subset(os.path.join(\"/kaggle/input/asl-signs\", row[1].path))\n    x = feature_converter(tf.convert_to_tensor(x)).cpu().numpy()\n    return x, row[1].label\n\ndef convert_and_save_data():\n    df = pd.read_csv(TRAIN_FILE)\n    df['label'] = df['sign'].map(label_map)\n    total = df.shape[0]\n    if QUICK_TEST:\n        total = QUICK_LIMIT\n    npdata = np.zeros((total, INPUT_SHAPE[0]*INPUT_SHAPE[1] + (SEGMENTS+1)*INPUT_SHAPE[1]*2))\n    nplabels = np.zeros(total)\n    for i, row in tqdm(enumerate(df.iterrows()), total=total):\n        (x,y) = convert_row(row)\n        npdata[i,:] = x\n        nplabels[i] = y\n        if QUICK_TEST and i == QUICK_LIMIT - 1:\n            break\n    \n    np.save(\"feature_data.npy\", npdata)\n    np.save(\"feature_labels.npy\", nplabels)\n    \n\ndef right_hand_percentage(x):\n    right = tf.gather(x, right_hand_landmarks, axis=1)\n    left = tf.gather(x, left_hand_landmarks, axis=1)\n    right_count = tf.reduce_sum(tf.where(tf.math.is_nan(right), tf.zeros_like(right), tf.ones_like(right)))\n    left_count = tf.reduce_sum(tf.where(tf.math.is_nan(left), tf.zeros_like(left), tf.ones_like(left)))\n    return right_count / (left_count+right_count)\n\n\n# Data Functions ###################################################################################\n\n\ndef prepare_main_csv(seed, num_splits, csv_path=CSV_PATH):\n    if Path(f\"{csv_path}/extended_new.csv\").exists():\n        data_csv = pd.read_csv(f\"{csv_path}/extended_new.csv\").reset_index(drop=True)\n    else:\n        data_csv = pd.read_csv(f\"{csv_path}/extended_train.csv\").reset_index(drop=True)\n\n        json_data = read_json_file()\n        json_df = pd.DataFrame.from_dict(json_data, orient=\"index\")\n        json_df = json_df.reset_index()\n        json_df.rename(columns={\"index\": \"sign\", 0: \"sign_val\"}, inplace=True)\n\n        data_csv = pd.merge(data_csv, json_df, on=\"sign\")\n        data_csv = data_csv.sample(frac=1.0, random_state=seed).reset_index(drop=True)\n        data_csv[\"hand\"] = data_csv[\"participant_id\"]\n        data_csv = data_csv.replace({\"hand\": di})\n        data_csv[\"fold_split\"] = (\n            data_csv[\"hand\"].astype(\"str\") + \"_\" + data_csv[\"sign_val\"].astype(\"str\")\n        )\n\n        # Splitting the data based on (hand & sign), grouped by participant_id\n        skf = StratifiedGroupKFold(n_splits=num_splits)\n        data_csv[\"fold\"] = -1\n        print(\"Splitting Data ------------------------------------>\")\n        for i, (train_index, test_index) in enumerate(\n            skf.split(data_csv.index, data_csv.fold_split, data_csv.participant_id)\n        ):\n            data_csv.loc[test_index, \"fold\"] = i\n            print(f\"fold {i} --> {len(test_index)}\")\n            print(data_csv.loc[test_index].participant_id.value_counts())\n        \n        data_csv.to_csv(f\"{csv_path}/extended_new.csv\", index=False)\n\n    return data_csv\n\n\ndef get_data(fold_num, cfg):\n    # Data Loading\n    print(\"Data Loading ----->\")\n    # train_x_full = np.load(f\"{ROOT_PATH}23_nonorm_feature_data.npy\").astype(np.float32)\n    # train_y_full = np.load(f\"{ROOT_PATH}23_nonorm_feature_labels.npy\").astype(np.uint8)\n    train_x_full = np.load(f\"{ROOT_PATH}feature_data.npy\").astype(np.float32)\n    train_y_full = np.load(f\"{ROOT_PATH}feature_labels.npy\").astype(np.uint8)\n\n    print(train_x_full.shape, train_y_full.shape)\n\n    if cfg['FLAG_DROP_Z']:\n        train_x_full = np.reshape(train_x_full, [train_x_full.shape[0], -1, 3])\n        train_x_full = train_x_full[:, :, 0:2]\n        train_x_full = np.reshape(train_x_full, [train_x_full.shape[0], -1])\n        print(train_x_full.shape, train_y_full.shape)\n\n    # Remove it with stratifiedkfold\n    train_df = prepare_main_csv(cfg['SEED'], cfg['NUM_SPLITS'])\n    train_idxs = train_df.index[train_df.fold != fold_num].to_numpy()\n    val_idxs = train_df.index[train_df.fold == fold_num].to_numpy()\n\n    train_x, train_y = train_x_full[train_idxs], train_y_full[train_idxs]\n    val_x, val_y = train_x_full[val_idxs], train_y_full[val_idxs]\n\n    del train_x_full, train_y_full, train_idxs, val_idxs\n\n    print(train_df[train_df.sequence_id == 1004211348])\n    return train_df, train_x, train_y, val_x, val_y\n\n\n# Utils ###################################################################################\n\ndef get_input_shape(num_frames, landmarks, flag_drop_z):\n    input_shape = (num_frames, landmarks * 3)\n\n    if flag_drop_z:\n        num_coords = 2\n    else:\n        num_coords = 3\n\n    return (num_frames, landmarks * num_coords)\n\n\ndef seed_it_all(seed=42):\n    \"\"\"Attempt to be Reproducible\"\"\"\n    tf.keras.backend.clear_session()\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n    tf.keras.utils.set_random_seed(seed)\n    tf.config.experimental.enable_op_determinism()\n\n\ndef read_json_file(file_path=f\"{DATA_DIR}/sign_to_prediction_index_map.json\"):\n    with open(file_path, \"r\") as file:\n        json_data = json.load(file)\n    return json_data\n\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = [\"x\", \"y\", \"z\"]\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)\n\n\ndef load_npz(f):\n    data = np.load(f)\n    return data[\"data\"]\n\n\ndef uniform_soup():\n    soups = []\n    ## Instantiating model\n\n    tf.keras.backend.clear_session()\n    model = get_model()\n    model_paths = Path(SAVE_DIR).glob(\"*.h5\")\n\n    ## Iterating Over all models\n    for path in tqdm(model_paths):\n        ## loading model wieghts\n        print(f\"### Loading {path}\")\n        model.load_weights(str(path))\n\n        ## Adding model weights in soup list\n        soup = [np.array(weights) for weights in model.weights]\n        soups.append(soup)\n\n    ## Averaing all weights\n    mean_soup = np.array(soups).mean(axis=0)\n\n    ## Replacing model's weight with Unifrom Soup Weights\n    for w1, w2 in zip(model.weights, mean_soup):\n        tf.keras.backend.set_value(w1, w2)\n\n    model.save_weights(f\"{SAVE_DIR}/uniform_soup.h5\")\n\n\nclass TFLiteModel(tf.Module):\n    \"\"\"\n    TensorFlow Lite model that takes input tensors and applies:\n        – a preprocessing model\n        – the ASL model\n    \"\"\"\n\n    def __init__(self, asl_model):\n        \"\"\"\n        Initializes the TFLiteModel with the specified feature generation model and main model.\n        \"\"\"\n        super(TFLiteModel, self).__init__()\n\n        # Load the feature generation and main models\n        self.prep_inputs = FeatureGen()\n        self.asl_model = asl_model\n\n    @tf.function(\n        input_signature=[\n            tf.TensorSpec(shape=[None, 543, 3], dtype=tf.float32, name=\"inputs\")\n        ]\n    )\n    def __call__(self, inputs):\n        \"\"\"\n        Applies the feature generation model and main model to the input tensors.\n\n        Args:\n            inputs: Input tensor with shape [batch_size, 543, 3].\n\n        Returns:\n            A dictionary with a single key 'outputs' and corresponding output tensor.\n        \"\"\"\n        x = self.prep_inputs(tf.cast(inputs, dtype=tf.float32))\n        outputs = self.asl_model(x)[0, :]\n\n        # Return a dictionary with the output tensor\n        return {\"outputs\": outputs}\n\n\n# Metrics ###################################################################################\n\nlabel_ls = list(range(0, 250))\nAVG_TYPE = \"weighted\"\n\n\ndef get_metrics(labels, preds_class):\n    metric_dict = {\n        \"accuracy\": metrics.accuracy_score(labels, preds_class),\n        \"f1_score\": metrics.f1_score(\n            labels, preds_class, labels=label_ls, zero_division=0, average=AVG_TYPE\n        ),\n        \"precision\": metrics.precision_score(\n            labels, preds_class, labels=label_ls, zero_division=0, average=AVG_TYPE\n        ),\n        \"recall\": metrics.recall_score(\n            labels, preds_class, labels=label_ls, zero_division=0, average=AVG_TYPE\n        ),\n        # 'roc': metrics.roc_auc_score(labels, preds_softmax, average='macro', multi_class='ovo', labels=label_ls),\n    }\n    for k, v in metric_dict.items():\n        print(k, v)\n\n\ndef compute_evaluation_metrics(model, data_x, data_y, decoder):\n    \"\"\"\n    Computes the evaluation metrics for the given model on the given data and prints classwise confusion matrix.\n\n    Args:\n    - model: The trained model to evaluate.\n    - data_x: The input data to evaluate the model on.\n    - data_y: The target data to evaluate the model on.\n    - decoder: A function to decode the model's output into readable text.\n    \"\"\"\n    # Compute the predicted classes and confusion matrix\n    batch_size = 1024\n    y_pred = model.predict(data_x, batch_size=1024)\n    print(y_pred.shape)\n    y_pred_classes = tf.cast(np.argmax(y_pred, axis=1), tf.uint8)\n    confusion_mtx = tf.math.confusion_matrix(data_y, y_pred_classes)\n\n    # Compute the evaluation metrics by class\n    num_classes = confusion_mtx.shape[0]\n    classwise_performance = {}\n    for i in range(num_classes):\n        tp = confusion_mtx[i, i]\n        fp = tf.reduce_sum(confusion_mtx[:, i]) - tp\n        fn = tf.reduce_sum(confusion_mtx[i, :]) - tp\n        tn = tf.reduce_sum(confusion_mtx[i]) - (tp - fp - fn)\n\n        classwise_performance[i] = dict(\n            accuracy=(tp + tn) / (tp + fp + tn + fn),\n            precision=tp / (tp + fp),\n            recall=tp / (tp + fn),\n        )\n        classwise_performance[i][\"f1_score\"] = (\n            2\n            * (\n                classwise_performance[i][\"precision\"]\n                * classwise_performance[i][\"recall\"]\n            )\n            / (\n                classwise_performance[i][\"precision\"]\n                + classwise_performance[i][\"recall\"]\n            )\n        )\n\n    # Sort the classwise performance by f1_score and print the results\n    classwise_performance = dict(\n        sorted(\n            classwise_performance.items(), key=lambda x: x[1][\"f1_score\"], reverse=True\n        )\n    )\n    print(\"\\n\\n... CLASSWISE CONFUSION MATRIX... \\n\")\n    for i, perf in classwise_performance.items():\n        print(\n            f\"Class {i:<3}  ({decoder[i]:^13})  -->  Accuracy: {perf['accuracy']:.2f}, Precision: {perf['precision']:.2f}, Recall: {perf['recall']:.2f}, F1 Score: {perf['f1_score']:.2f}\"\n        )\n\n\n# Model Utils ################################################################################\n\noutput_bias = tf.keras.initializers.Constant(1.0 / 250.0)\n\n\nclass MSD(tf.keras.layers.Layer):\n    def __init__(\n        self,\n        units,\n        fold_num,\n        cfg,\n        **kwargs,\n    ):\n        super().__init__(**kwargs)\n\n        self.lin = tf.keras.layers.Dense(\n            units,\n            activation=None,\n            use_bias=True,\n            bias_initializer=output_bias,\n            # kernel_regularizer=R.l2(WEIGHT_REGULARIZE)\n        )\n\n        rate_dropout = cfg[\"MSD_DROPOUT\"]\n        if cfg[\"MSD_DROP_TYPE\"] == \"normal\":\n            self.dropouts = [\n                tf.keras.layers.Dropout((rate_dropout - 0.2), seed=135 + fold_num),\n                tf.keras.layers.Dropout((rate_dropout - 0.1), seed=690 + fold_num),\n                tf.keras.layers.Dropout((rate_dropout), seed=275 + fold_num),\n                tf.keras.layers.Dropout((rate_dropout + 0.1), seed=348 + fold_num),\n                tf.keras.layers.Dropout((rate_dropout + 0.2), seed=861 + fold_num),\n            ]\n\n        elif cfg[\"MSD_DROP_TYPE\"] == \"gaussian\":\n            self.dropouts = [\n                tf.keras.layers.GaussianDropout((rate_dropout - 0.2)),\n                tf.keras.layers.GaussianDropout((rate_dropout - 0.1)),\n                tf.keras.layers.GaussianDropout(rate_dropout),\n                tf.keras.layers.GaussianDropout((rate_dropout + 0.1)),\n                tf.keras.layers.GaussianDropout((rate_dropout + 0.2)),\n            ]\n\n    def call(self, inputs):\n        for ii, drop in enumerate(self.dropouts):\n            if ii == 0:\n                out = self.lin(drop(inputs)) / 5.0\n            else:\n                out += self.lin(drop(inputs)) / 5.0\n        return out\n\n\nclass ResidualBlock(tf.keras.layers.Layer):\n    def __init__(self, units, dropout):\n        super().__init__()\n        self.linear = tf.keras.layers.Dense(units)\n        self.bn = tf.keras.layers.BatchNormalization()\n        self.act = tf.keras.layers.Activation(\"gelu\")\n        if dropout != 0:\n            self.drop = tf.keras.layers.Dropout(dropout)\n            self.flag_use_drop = True\n        else:\n            self.flag_use_drop = False\n\n    def call(self, x):\n        x = self.linear(x)\n        x = self.bn(x)\n        x = self.act(x)\n        if self.flag_use_drop:\n            x = self.drop(x)\n        return x\n\n\nclass GRUModel(tf.keras.layers.Layer):\n    def __init__(self, units, dropout, num_blocks):\n        super().__init__()\n        self.start_gru = tf.keras.layers.GRU(\n            units=units, dropout=0.0, return_sequences=True\n        )\n        self.end_gru = tf.keras.layers.GRU(\n            units=units, dropout=dropout, return_sequences=False\n        )\n\n        if (num_blocks - 2) > 0:\n            self.gru_blocks = [\n                tf.keras.layers.GRU(units=units, dropout=dropout, return_sequences=True)\n                * (num_blocks - 2)\n            ]\n            self.flag_use_gru_blocks = True\n        else:\n            self.flag_use_gru_blocks = False\n\n    def call(self, x):\n        x = self.start_gru(x)\n        if self.flag_use_gru_blocks:\n            for blk in self.gru_blocks:\n                x = blk(x)\n        x = self.end_gru(x)\n        return x\n\n\ndef model_utils(cfg, fold_num):\n    metric_ls = [\n        tf.keras.metrics.SparseCategoricalAccuracy(),\n        tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5),\n    ]\n\n    cb_list = [\n        tf.keras.callbacks.EarlyStopping(\n            patience=5,\n            restore_best_weights=True,\n            verbose=1,\n            monitor=cfg[\"TARGET_METRIC\"],\n        ),\n        tf.keras.callbacks.ReduceLROnPlateau(patience=2, factor=0.8, verbose=1),\n        tf.keras.callbacks.ModelCheckpoint(\n            f\"{SAVE_DIR}/best_acc_{fold_num}.h5\",\n            monitor=cfg[\"TARGET_METRIC\"],\n            verbose=0,\n            save_best_only=True,\n            save_weights_only=True,\n            mode=\"max\",\n            save_freq=\"epoch\",\n        ),\n    ]\n\n    if cfg[\"FLAG_WANDB\"]:\n        cb_list += [#WandbMetricsLogger()\n            WandbCallback(\n                monitor=cfg[\"TARGET_METRIC\"],\n                log_weights=False,\n                log_evaluation=False,\n                save_model=False,\n            )\n        ]\n\n    opt = tfa.optimizers.AdamW(weight_decay=0, learning_rate=cfg[\"LR\"])\n    # opt = tf.keras.optimizers.Adam(learning_rate=LR)\n    # opt = tfa.optimizers.RectifiedAdam(learning_rate=LR)\n    # opt = tfa.optimizers.Lookahead(opt, sync_period=5)\n\n    return metric_ls, cb_list, opt\n\n\n######################################################################################################################\n\nlip_landmarks = [\n    61,\n    185,\n    40,\n    39,\n    37,\n    0,\n    267,\n    269,\n    270,\n    409,\n    291,\n    146,\n    91,\n    181,\n    84,\n    17,\n    314,\n    405,\n    321,\n    375,\n    78,\n    191,\n    80,\n    81,\n    82,\n    13,\n    312,\n    311,\n    310,\n    415,\n    95,\n    88,\n    178,\n    87,\n    14,\n    317,\n    402,\n    318,\n    324,\n    308,\n]\n\n# Analyzing Handedness\nleft_handed_signer = [\n    16069,\n    32319,\n    36257,\n    22343,\n    27610,\n    61333,\n    34503,\n    55372,\n    37055,\n]  # both_hands_signer-> 37055\nright_handed_signer = [\n    26734,\n    28656,\n    25571,\n    62590,\n    29302,\n    49445,\n    53618,\n    18796,\n    4718,\n    2044,\n    37779,\n    30680,\n]\nlip_landmarks = [\n    61,\n    185,\n    40,\n    39,\n    37,\n    0,\n    267,\n    269,\n    270,\n    409,\n    291,\n    146,\n    91,\n    181,\n    84,\n    17,\n    314,\n    405,\n    321,\n    375,\n    78,\n    191,\n    80,\n    81,\n    82,\n    13,\n    312,\n    311,\n    310,\n    415,\n    95,\n    88,\n    178,\n    87,\n    14,\n    317,\n    402,\n    318,\n    324,\n    308,\n]\n\ndi = {}\nfor k in left_handed_signer:\n    di[k] = 0\nfor k in right_handed_signer:\n    di[k] = 1\n\nleft_hand_landmarks = list(range(468, 468 + 21))\nright_hand_landmarks = list(range(522, 522 + 21))\n\naveraging_sets = [\n    [0, 468],\n    [489, 33],\n]  ## average over the entire face, and the entire 'pose'\n\npoint_landmarks = [\n    item\n    for sublist in [lip_landmarks, left_hand_landmarks, right_hand_landmarks]\n    for item in sublist\n]\n\nLANDMARKS = len(point_landmarks) #+ len(averaging_sets)\n\n# Fixed  ##################################################################################\n\nFLAG_DROP_Z = False\nROWS_PER_FRAME = 543\nNUM_FRAMES = 15\nINPUT_SHAPE = get_input_shape(NUM_FRAMES, LANDMARKS, FLAG_DROP_Z)\nSEGMENTS = 3\nNUM_BASE_FEATS = (SEGMENTS + 1) * INPUT_SHAPE[1] * 2\nFLAT_FRAME_SHAPE = NUM_BASE_FEATS + (INPUT_SHAPE[0] * INPUT_SHAPE[1])\ndecoder = {v: k for k, v in read_json_file().items()}","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:05:58.266117Z","iopub.execute_input":"2023-04-30T15:05:58.266504Z","iopub.status.idle":"2023-04-30T15:05:58.290727Z","shell.execute_reply.started":"2023-04-30T15:05:58.266465Z","shell.execute_reply":"2023-04-30T15:05:58.289296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile trainer.py\n\nimport gc\nimport os\nimport pprint\nimport warnings\nimport wandb\nimport hydra\nfrom omegaconf import DictConfig\nfrom zipfile import ZipFile\ntry:\n    import tflite_runtime.interpreter as tflite\n    FLAG_INTERPRET = True\nexcept:\n    FLAG_INTERPRET = False\n    print(\"TFlite Interpretation not possible\")\nwarnings.filterwarnings(\"ignore\")\nos.environ[\"TF_DETERMINISTIC_OPS\"] = \"1\"\nos.environ[\"TF_CUDNN_DETERMINISTIC\"] = \"1\"\nos.environ[\"TF_CPP_MIN_LOG_LEVEL\"] = \"2\"\nimport time\nimport numpy as np\nimport tensorflow as tf\nfrom common_func import *\n\nseed_it_all()\nstart_time = time.time()\n\n# Flags  ##################################################################################\nif False:\n    mixed_precision.set_global_policy(\"mixed_float16\")\n    tf.config.optimizer.set_jit(True)\n\n# Model  ###################################################################################\n\n\ndef get_model(\n    cfg,\n    fold_num=0,\n    n_labels=250,\n    flat_frame_len=FLAT_FRAME_SHAPE,\n    flag_model_summary=False,\n    flag_with_cb_list=False,\n):\n    print(\"Model Loading ----->\")\n    _inputs = tf.keras.layers.Input(shape=(flat_frame_len,))\n\n    # import ipdb\n    # ipdb.set_trace()\n    x = _inputs[:, :NUM_BASE_FEATS]\n    x_conv = tf.reshape(_inputs[:, NUM_BASE_FEATS:], (-1, NUM_FRAMES, INPUT_SHAPE[1]))\n\n    # Concat Dilated Convolutions with actual data\n    gru_out = GRUModel(\n        cfg[\"NUM_GRU_UNITS\"], cfg[\"RATE_GRU_DROPOUT\"], cfg[\"NUM_GRU_BLOCKS\"]\n    )(x_conv)\n    \n    if cfg['FLAG_CONCAT_FEATS']:\n        x = tf.keras.layers.concatenate([gru_out, x], axis=1)\n    else:\n        x = gru_out\n    print(\"Concatenate Shape\", x.shape)\n\n    # Residual Block\n    x = ResidualBlock(cfg[\"NUM_RESIDUAL_UNITS\"], 0.25)(x)\n    x += ResidualBlock(cfg[\"NUM_RESIDUAL_UNITS\"], 0.0)(x)\n\n    # Final output MSD Layer\n    x = MSD(units=n_labels, fold_num=fold_num, cfg=cfg)(x)\n    _outputs = tf.keras.layers.Softmax(dtype=\"float32\")(x)\n\n    # Build the model\n    model = tf.keras.models.Model(inputs=_inputs, outputs=_outputs)\n    metric_ls, cb_list, opt = model_utils(cfg, fold_num)\n    model.compile(opt, \"sparse_categorical_crossentropy\", metrics=metric_ls)\n\n    if flag_model_summary:\n        model.summary()\n\n    if flag_with_cb_list:\n        return model, cb_list\n    else:\n        return model\n\n\ndef tflite_conversion(model):    \n    # TFLite Conversion\n    tflite_keras_model = TFLiteModel(model)\n    demo_output = tflite_keras_model(load_relevant_data_subset('1004211348.parquet'))[\"outputs\"]\n    decoder[np.argmax(demo_output.numpy(), axis=-1)]\n\n    keras_model_converter = tf.lite.TFLiteConverter.from_keras_model(tflite_keras_model)\n    tflite_model = keras_model_converter.convert()\n    \n    tf_lite_model_path = f'{SAVE_DIR}/model.tflite'\n    with open(tf_lite_model_path, 'wb') as f:\n        f.write(tflite_model)\n        \n    ZipFile('submission.zip', mode='w').write(tf_lite_model_path)\n\n    if FLAG_INTERPRET:\n        interpreter = tflite.Interpreter(tf_lite_model_path)\n        found_signatures = list(interpreter.get_signature_list().keys())\n        prediction_fn = interpreter.get_signature_runner(\"serving_default\")\n        output = prediction_fn(inputs=load_relevant_data_subset('1004211348.parquet'))\n        sign = np.argmax(output[\"outputs\"])\n\n        print(\"PRED : \", decoder[sign])\n        # print(\"GT   : \", train_df.sign[0])\n\n    \n@hydra.main(version_base=None, config_path=\"./\", config_name=\"config\")\ndef my_app(cfg: DictConfig):\n    print(\"*\" * 75)\n    config = dict(cfg[\"CFG\"])\n    seed_it_all(config[\"SEED\"])\n    config[\"FLAG_DROP_Z\"] = FLAG_DROP_Z\n    if config[\"FLAG_DEBUG\"]:\n        config[\"NUM_EPOCHS\"] = 3\n        config[\"FLAG_WANDB\"] = config[\"FLAG_WANDB\"] and False\n    else:\n        config[\"FLAG_WANDB\"] = config[\"FLAG_WANDB\"] and True\n\n    pprint.pprint(config)    \n    true = np.array([])\n    oof = np.array([])\n    for fold_cnt, fold_num in enumerate(\n        range(config[\"FOLD_START\"], config[\"FOLD_END\"]+1)\n    ):\n        print(\"#\" * 25)\n        print(f\"### Fold {fold_num}\")\n        if config[\"FLAG_WANDB\"]:\n            wandb.init(project=\"isle_analysis\", group=config[\"DESCRIPTION\"])\n\n        seed_it_all(config[\"SEED\"] + fold_num)\n        train_df, train_x, train_y, val_x, val_y = get_data(\n            fold_num, config\n        )\n\n        model, cb_list = get_model(\n            config, fold_num, flag_with_cb_list=True, flag_model_summary=(fold_cnt == 0)\n        )\n\n        history = model.fit(\n            train_x,\n            train_y,\n            validation_data=(val_x, val_y),\n            verbose=2,\n            epochs=config[\"NUM_EPOCHS\"],\n            callbacks=cb_list,\n            batch_size=config[\"BATCH_SIZE\"],\n            workers=8,\n        )\n\n        oof_p = model.predict(val_x, batch_size=config[\"BATCH_SIZE\"], verbose=2)\n        oof_p = np.argmax(oof_p, axis=1)\n        true = np.concatenate([true, val_y])\n        oof = np.concatenate([oof, oof_p])\n\n        print(\"#\" * 25)\n        print(f\"### Evaluation Metrics\")\n        model.evaluate(val_x, val_y)\n\n        if fold_num == config[\"FOLD_END\"]:\n            compute_evaluation_metrics(model, val_x, val_y, decoder=decoder)\n            \n        del train_df, train_x, train_y, val_x, val_y\n        gc.collect()\n\n        if config[\"FLAG_WANDB\"]:\n            wandb.finish()\n            \n    # PRINT OVERALL RESULTS\n    print(\"#\" * 25)\n    print(f\"Overall Metrics\")\n    get_metrics(true, oof)\n    tf.keras.backend.clear_session()\n\n    if config[\"FLAG_GEN_TFLITE\"]:\n        tflite_conversion(model)\n        \n    \nif __name__ == \"__main__\":\n    cfg = my_app()\n    print(\"Total Time: \", time.time() - start_time)\n\n    # Dilated Convolutions\n    # conv_1 = tf.keras.layers.Conv1D(5, 1, strides=1, activation='silu')(x_conv)\n    # conv_3 = tf.keras.layers.Conv1D(5, 1, strides=3, activation='silu')(x_conv)\n    # conv_5 = tf.keras.layers.Conv1D(5, 1, strides=5, activation='silu')(x_conv)\n    # conv_15 = tf.keras.layers.Conv1D(5, 1, strides=15, activation='silu')(x_conv)\n    # conv_out = tf.keras.layers.concatenate([conv_1, conv_3, conv_5, conv_15], axis=1)\n    # conv_out = tf.reshape(conv_out, (-1, conv_out.shape[1] * conv_out.shape[2]))\n%%writefile config.yaml\nCFG:\n    # General Params  ##########################################################################\n\n    DESCRIPTION: initial trials\n    LR: 6e-4\n    BATCH_SIZE: 512 # 512\n    NUM_EPOCHS: 100\n    NUM_SPLITS: 7\n    TARGET_METRIC: \"val_sparse_categorical_accuracy\"\n\n    FOLD_START: 1\n    FOLD_END: 1\n    \n    SEED: 42\n    \n    # Flags  ##################################################################################\n    \n    FLAG_DROP_Z: False\n    FLAG_GEN_TFLITE: True\n#     FLAG_GEN_TFLITE: False\n    \n    FLAG_CONCAT_FEATS: False # Use mean and standard deviation information captured in the dataset\n\n\n    FLAG_DEBUG: False\n#     FLAG_WANDB: True\n    FLAG_WANDB: False\n    # FLAG_DEBUG: True\n\n    # Model  ##################################################################################\n\n    # NUM_RESIDUAL_UNITS: 128\n    NUM_RESIDUAL_UNITS: 1024\n\n    # NUM_GRU_UNITS: 128\n    NUM_GRU_UNITS: 512\n    RATE_GRU_DROPOUT: 0.5\n    NUM_GRU_BLOCKS: 1 # 2 minimal value\n\n    MSD_DROP_TYPE: \"normal\"\n    MSD_DROPOUT: 0.5","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:05:58.293015Z","iopub.execute_input":"2023-04-30T15:05:58.293831Z","iopub.status.idle":"2023-04-30T15:05:58.314812Z","shell.execute_reply.started":"2023-04-30T15:05:58.293766Z","shell.execute_reply":"2023-04-30T15:05:58.313165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile config.yaml\n!python3 trainer.py","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:05:58.316874Z","iopub.execute_input":"2023-04-30T15:05:58.317318Z","iopub.status.idle":"2023-04-30T15:05:58.333828Z","shell.execute_reply.started":"2023-04-30T15:05:58.317282Z","shell.execute_reply":"2023-04-30T15:05:58.332291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"\\n... PIP INSTALLS STARTING ...\\n\")\n!pip install tflite-runtime\nimport tflite_runtime.interpreter as tflite\nprint(\"\\n... PIP INSTALLS COMPLETE ...\\n\")\n\n\nprint(\"\\n... IMPORTS STARTING ...\\n\")\nprint(\"\\n\\tVERSION INFORMATION\")\n\n# Machine Learning and Data Science Imports (basics)\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport tensorflow_io as tfio; print(f\"\\t\\t– TENSORFLOW-IO VERSION: {tfio.__version__}\");\nimport tensorflow_addons as tfa; print(f\"\\t\\t– TENSORFLOW-ADDONS VERSION: {tfa.__version__}\");\nimport pandas as pd; pd.options.mode.chained_assignment = None; pd.set_option('display.max_columns', None);\nimport numpy as np; print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\");\nimport sklearn; print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\");\n\n# Built-In Imports (mostly don't worry about these)\nfrom sklearn.model_selection import StratifiedKFold, StratifiedGroupKFold\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom zipfile import ZipFile\nfrom glob import glob\nimport Levenshtein\nimport warnings\nimport requests\nimport hashlib\nimport imageio\nimport IPython\nimport sklearn\nimport urllib\nimport zipfile\nimport pickle\nimport random\nimport shutil\nimport string\nimport json\nimport math\nimport time\nimport gzip\nimport ast\nimport sys\nimport io\nimport os\nimport gc\nimport re\n\n# Visualization Imports (overkill)\nfrom matplotlib.animation import FuncAnimation\nfrom matplotlib.colors import ListedColormap\nfrom matplotlib.patches import Rectangle\nimport matplotlib.patches as patches\nimport plotly.graph_objects as go\nfrom IPython.display import HTML\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm; tqdm.pandas();\nimport plotly.express as px\nimport tifffile as tif\nimport seaborn as sns\nfrom PIL import Image, ImageEnhance; Image.MAX_IMAGE_PIXELS = 5_000_000_000;\nimport matplotlib; print(f\"\\t\\t– MATPLOTLIB VERSION: {matplotlib.__version__}\");\nfrom matplotlib import animation, rc; rc('animation', html='jshtml')\nimport plotly\nimport PIL\nimport cv2\n\nimport plotly.io as pio\nprint(pio.renderers)\n\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\nseed_it_all()\n\nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:05:58.335786Z","iopub.execute_input":"2023-04-30T15:05:58.336273Z","iopub.status.idle":"2023-04-30T15:08:28.476668Z","shell.execute_reply.started":"2023-04-30T15:05:58.336236Z","shell.execute_reply":"2023-04-30T15:08:28.473015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def flatten_l_o_l(nested_list):\n    \"\"\"Flatten a list of lists into a single list.\n\n    Args:\n        nested_list (list): \n            – A list of lists (or iterables) to be flattened.\n\n    Returns:\n        list: A flattened list containing all items from the input list of lists.\n    \"\"\"\n    return [item for sublist in nested_list for item in sublist]\n\n\ndef print_ln(symbol=\"-\", line_len=110, newline_before=False, newline_after=False):\n    \"\"\"Print a horizontal line of a specified length and symbol.\n\n    Args:\n        symbol (str, optional): \n            – The symbol to use for the horizontal line\n        line_len (int, optional): \n            – The length of the horizontal line in characters\n        newline_before (bool, optional): \n            – Whether to print a newline character before the line\n        newline_after (bool, optional): \n            – Whether to print a newline character after the line\n    \"\"\"\n    if newline_before: print();\n    print(symbol * line_len)\n    if newline_after: print();\n        \n        \ndef read_json_file(file_path):\n    \"\"\"Read a JSON file and parse it into a Python object.\n\n    Args:\n        file_path (str): The path to the JSON file to read.\n\n    Returns:\n        dict: A dictionary object representing the JSON data.\n        \n    Raises:\n        FileNotFoundError: If the specified file path does not exist.\n        ValueError: If the specified file path does not contain valid JSON data.\n    \"\"\"\n    try:\n        # Open the file and load the JSON data into a Python object\n        with open(file_path, 'r') as file:\n            json_data = json.load(file)\n        return json_data\n    except FileNotFoundError:\n        # Raise an error if the file path does not exist\n        raise FileNotFoundError(f\"File not found: {file_path}\")\n    except ValueError:\n        # Raise an error if the file does not contain valid JSON data\n        raise ValueError(f\"Invalid JSON data in file: {file_path}\")\n        \ndef get_sign_df(pq_path, invert_y=True):\n    sign_df = pd.read_parquet(pq_path)\n    \n    # y value is inverted (Thanks @danielpeshkov)\n    if invert_y: sign_df[\"y\"] *= -1 \n        \n    return sign_df\n\nROWS_PER_FRAME = 543  # number of landmarks per frame\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:45:29.136760Z","iopub.execute_input":"2023-04-30T15:45:29.137396Z","iopub.status.idle":"2023-04-30T15:45:29.155549Z","shell.execute_reply.started":"2023-04-30T15:45:29.137336Z","shell.execute_reply":"2023-04-30T15:45:29.154006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR         = \"/kaggle/input/asl-signs\"\nEXTEND_TRAIN_DIR = \"/kaggle/input/gislr-extended-train-dataframe\" \nNP_FILE_DIR      = \"/kaggle/input/isolated-sign-language-aggregation-preparation\"\n\nprint(\"\\n... BASIC DATA SETUP STARTING ...\\n\")\nprint(\"\\n\\n... LOAD TRAIN DATAFRAME FROM CSV FILE ...\\n\")\n\nLOAD_EXTENDED = True\nif LOAD_EXTENDED and os.path.isfile(os.path.join(EXTEND_TRAIN_DIR, \"extended_train.csv\")):\n    train_df = pd.read_csv(os.path.join(EXTEND_TRAIN_DIR, \"extended_train.csv\"))\nelse:\n    train_df = pd.read_csv(os.path.join(DATA_DIR, \"train.csv\"))\n    train_df[\"path\"] = DATA_DIR+\"/\"+train_df[\"path\"]\ndisplay(train_df)\n\nprint(\"\\n\\n... LOAD SIGN TO PREDICTION INDEX MAP FROM JSON FILE ...\\n\")\ns2p_map = {k.lower():v for k,v in read_json_file(os.path.join(DATA_DIR, \"sign_to_prediction_index_map.json\")).items()}\np2s_map = {v:k for k,v in read_json_file(os.path.join(DATA_DIR, \"sign_to_prediction_index_map.json\")).items()}\nencoder = lambda x: s2p_map.get(x.lower())\ndecoder = lambda x: p2s_map.get(x)\n\nDEMO_ROW = 283\nprint(f\"\\n\\n... DEMO SIGN/EVENT DATAFRAME FOR ROW {DEMO_ROW} - SIGN={train_df.iloc[DEMO_ROW]['sign']} ...\\n\")\ndemo_sign_df = get_sign_df(train_df.iloc[DEMO_ROW][\"path\"])\ndisplay(demo_sign_df)\n\n# Landmark IDs start at 0 for each respective type and count up\nFRAME_TYPE_ORDER_DETAIL = demo_sign_df.groupby(\"frame\")[\"type\"].apply(list).values[0]\nFRAME_TYPE_ORDER = sorted(set(FRAME_TYPE_ORDER_DETAIL))\nprint(FRAME_TYPE_ORDER)\n\n# https://www.kaggle.com/competitions/asl-signs/discussion/391812#2168354\nlipsUpperOuter = [61, 185, 40, 39, 37, 0, 267, 269, 270, 409, 291]\nlipsLowerOuter = [146, 91, 181, 84, 17, 314, 405, 321, 375, 291]\nlipsUpperInner = [78, 191, 80, 81, 82, 13, 312, 311, 310, 415, 308]\nlipsLowerInner = [78, 95, 88, 178, 87, 14, 317, 402, 318, 324, 308]\nlips = lipsUpperOuter + lipsLowerOuter + lipsUpperInner + lipsLowerInner\nFRAME_TYPE_IDX_MAP = {\n    \"lips\"       : np.array(lips),\n    \"left_hand\"  : np.arange(468, 489),\n    \"pose\"       : np.arange(489, 522),\n    \"right_hand\" : np.arange(522, 543),\n}\n\nfor k,v in FRAME_TYPE_IDX_MAP.items():\n    print(k, len(v))","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:45:32.658656Z","iopub.execute_input":"2023-04-30T15:45:32.659105Z","iopub.status.idle":"2023-04-30T15:45:33.356937Z","shell.execute_reply.started":"2023-04-30T15:45:32.659069Z","shell.execute_reply":"2023-04-30T15:45:33.355425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn.functional as F\nimport torch.nn as nn\n\n#num_landmark = 543\nmax_length = 80\nnum_class  = 250\nnum_point  = 82  # LIP, LHAND, RHAND\n\ndef pack_seq(\n    seq,\n):\n    length = [len(s) for s in seq]\n    batch_size = len(seq)\n    num_landmark=seq[0].shape[1]\n\n    x = torch.zeros((batch_size, max(length), num_landmark, 3)).to(seq[0].device)\n    x_mask = torch.zeros((batch_size, max(length))).to(seq[0].device)\n    for b in range(batch_size):\n        L = length[b]\n        x[b, :L] = seq[b][:L]\n        x_mask[b, L:] = 1\n    x_mask = (x_mask>0.5)\n    x = x.reshape(batch_size,-1,num_landmark*3)\n    return x, x_mask\n\n\nclass FeedForward(nn.Module):\n    def __init__(self, embed_dim, hidden_dim):\n        super().__init__()\n        self.mlp = nn.Sequential(\n            nn.Linear(embed_dim, hidden_dim),\n            nn.ReLU(inplace=True),\n            nn.Linear(hidden_dim, embed_dim),\n        )\n    def forward(self, x):\n        return self.mlp(x)\n\n\n#https://pytorch.org/docs/stable/generated/torch.nn.MultiheadAttention.html\nclass MultiHeadAttention(nn.Module):\n    def __init__(self,\n            embed_dim,\n            num_head,\n            batch_first,\n        ):\n        super().__init__()\n        self.mha = nn.MultiheadAttention(\n            embed_dim,\n            num_heads=num_head,\n            bias=True,\n            add_bias_kv=False,\n            kdim=None,\n            vdim=None,\n            dropout=0.0,\n            batch_first=batch_first,\n        )\n\n    def forward(self, x, x_mask):\n        out, _ = self.mha(x,x,x, key_padding_mask=x_mask)\n        return out\n\n\ndef positional_encoding(length, embed_dim):\n    dim = embed_dim//2\n\n    position = np.arange(length)[:, np.newaxis] \n    dim = np.arange(dim)[np.newaxis, :]/dim   # (1, dim)\n\n    angle = 1 / (10000**dim)         # (1, dim)\n    angle = position * angle    # (pos, dim)\n\n    pos_embed = np.concatenate(\n        [np.sin(angle), np.cos(angle)],\n        axis=-1\n    )\n    pos_embed = torch.from_numpy(pos_embed).float()\n    return pos_embed\n\nclass TransformerBlock(nn.Module):\n    def __init__(self,\n        embed_dim,\n        num_head,\n        out_dim,\n        batch_first=True,\n    ):\n        super().__init__()\n        self.attn  = MultiHeadAttention(embed_dim, num_head,batch_first)\n        self.ffn   = FeedForward(embed_dim, out_dim)\n        self.norm1 = nn.LayerNorm(embed_dim)\n        self.norm2 = nn.LayerNorm(out_dim)\n\n    def forward(self, x, x_mask=None):\n        x = x + self.attn((self.norm1(x)), x_mask)\n        x = x + self.ffn((self.norm2(x)))\n        return x\n\nclass Net(nn.Module):\n\n    def __init__(self, num_class=num_class):\n        super().__init__()\n        self.output_type = ['inference', 'loss']\n\n        num_block = 1\n        embed_dim = 1024\n        num_head  = 8\n\n        pos_embed = positional_encoding(max_length, embed_dim)\n        # self.register_buffer('pos_embed', pos_embed)\n        self.pos_embed = nn.Parameter(pos_embed)\n\n        self.cls_embed = nn.Parameter(torch.zeros((1, embed_dim)))\n        self.x_embed = nn.Sequential(\n            nn.Linear(num_point * 3, embed_dim, bias=False),\n        )\n\n        self.encoder = nn.ModuleList([\n            TransformerBlock(\n                embed_dim,\n                num_head,\n                embed_dim,\n            ) for i in range(num_block)\n        ])\n        self.logit = nn.Linear(embed_dim, num_class)\n\n    def forward(self, batch):\n        length = [len(x) for x in batch['xyz']]\n        xyz = batch['xyz']\n\n        x, x_mask = pack_seq(xyz)\n        B,L,_ = x.shape\n        x = self.x_embed(x)\n        x = x + self.pos_embed[:L].unsqueeze(0)\n\n        x = torch.cat([\n            self.cls_embed.unsqueeze(0).repeat(B,1,1),\n            x\n        ],1)\n        x_mask = torch.cat([\n            torch.zeros(B,1).to(x_mask),\n            x_mask\n        ],1)\n\n\n        #x = F.dropout(x,p=0.25,training=self.training)\n        for block in self.encoder:\n            x = block(x,x_mask)\n\n        cls = x[:,0]\n        cls = F.dropout(cls,p=0.4,training=self.training)\n        logit = self.logit(cls)\n\n        output = {}\n        if 'loss' in self.output_type:\n            output['label_loss'] = F.cross_entropy(logit, batch['label'])\n\n        if 'inference' in self.output_type:\n            output['sign'] = torch.softmax(logit,-1)\n\n        return output\n\n\ndef pre_process(xyz):\n    xyz = xyz - xyz[~torch.isnan(xyz)].mean(0,keepdims=True) #noramlisation to common mean\n    xyz = xyz / xyz[~torch.isnan(xyz)].std(0, keepdims=True)\n    \n    lip = xyz[:, LIP]\n    lhand = xyz[:, LHAND]\n    rhand = xyz[:, RHAND]\n    xyz = torch.cat([ #(none, 82, 3)\n        lip,\n        lhand,\n        rhand,\n    ],1)\n    xyz[torch.isnan(xyz)] = 0\n    xyz = xyz[:max_length]\n    return xyz","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"''' \nIMPORTANT:\nALL test code bellow written by jacob: do not use in normal runs yet\n'''\n\n\n'''Here is a basic approach we can classify as a naive improvment on the demo:'''\nimport torch\nimport torch.nn.functional as F\nimport torch.nn as nn\n\n#num_landmark = 543\nmax_length = 80\nnum_class  = 250\nnum_point  = 82  # LIP, LHAND, RHAND\n\nclass FeedForward(nn.Module): #FEEDFORWARD NOW CONTAINS BATCHNORM AND DROPOUT\n    def __init__(self, embed_dim, hidden_dim):\n        super().__init__()\n        self.mlp = nn.Sequential(\n           nn.Linear( embed_dim, hidden_dim ),\n            nn.BatchNorm1d(hidden_dim ),\n            nn.ReLU(inplace=True),\n            nn.Dropout(0.5),#Just assuming this is the default to go with im not actually sure.\n            nn.Linear(hidden_dim, embed_dim),\n            nn.BatchNorm1d(embed_dim)\n        )\n    def forward(self, x):\n        return self.mlp(x)\n\n\nclass Net(nn.Module):\n\n    def __init__(self, num_class=num_class):\n        super().__init__()\n        self.output_type = ['inference', 'loss']\n\n        num_block = 1\n        embed_dim = 1024\n        num_head  = 8\n\n        self.pos_embed = nn.Parameter(torch.zeros((max_length, embed_dim)))\n        nn.init.normal_(self.pos_embed, std=0.02)\n        self.cls_embed = nn.Parameter(torch.zeros((1, 1, embed_dim)))\n        nn.init.normal_(self.cls_embed, std=0.02)\n        self.x_embed = nn.Sequential(\n            nn.Linear(num_point * 3, embed_dim, bias=False),\n        )\n        \n        #USE THE NATIVE TRANSFORMERENCODER INSTEAD OF MAKING IT OURSELVES\n        self.transformer_encoder = nn.TransformerEncoder( \n            nn.TransformerEncoderLayer(\n                embed_dim,\n                num_head,\n                dim_feedforward=embed_dim\n            ),\n            num_layers=num_block\n        )\n        \n        self.linear_layer = nn.Sequential(\n            nn.Linear(embed_dim, 512),\n            nn.ReLU(inplace=True),#This saves us from needing to return it (read documentation)\n            nn.Linear(512, 256),\n            nn.ReLU(inplace=True),\n            nn.Linear(256, num_class)\n        )\n\n        self.dropout = nn.Dropout(0.2)\n\n    def forward(self, batch):\n        xyz = batch['xyz']\n\n        #Pack sequences (This is likely a pain-point) Had to find this code so may not work.\n        x = nn.utils.rnn.pack_sequence(xyz, enforce_sorted=False)\n        x, x_lengths = nn.utils.rnn.pad_packed_sequence(x) #Pads our tensors with 0's where needed.\n\n        #Positional encoding (This is likely a pain-point) Had to find this code so may not work.\n        x = x + self.pos_embed[: x.size(0), :].unsqueeze(1).to(x.device) \n        x = x + self.cls_embed.repeat(x.size(0), 1, 1)\n\n        #Embedding\n        x = self.x_embed(x.view(-1, num_point * 3)).view(x.size(0), x.size(1), -1)\n\n           \n        #Finally, Apply the layers.\n        x = self.transformer_encoder(x)\n        x = self.dropout(x)\n        x = self.linear_layer(x)\n        return x\n'''\nHere is the LTSM approach we cited in our proposal. Again i have no clue if this even works\n'''\nimport torch\nimport torch.nn.functional as F\nimport torch.nn as nn\n\n#num_landmark = 543\nmax_length = 80\nnum_class = 250\nnum_point = 82  # LIP, LHAND, RHAND\n\ndef pack_seq(seq): #COPPIED OVER FROM UP TOP SO WE DON'T HAVE TO RUN THAT CELL\n    length = [len(s) for s in seq]\n    batch_size = len(seq)\n    num_landmark = seq[0].shape[1]\n\n    x = torch.zeros((batch_size, max(length), num_landmark, 3)).to(seq[0].device)\n    x_mask = torch.zeros((batch_size, max(length))).to(seq[0].device)\n    for b in range(batch_size):\n        L = length[b]\n        x[b, :L] = seq[b][:L]\n        x_mask[b, L:] = 1\n    x_mask = (x_mask>0.5)\n    x = x.reshape(batch_size, -1, num_landmark*3)\n    return x, x_mask\n\n\nclass Net(nn.Module):\n\n    def __init__(self, num_class=num_class):\n        super().__init__()\n        self.output_type = ['inference', 'loss']\n\n        num_block = 1\n        lstm_hidden_size = 256\n        lstm_num_layers = 1\n        pos_embed_dim = 128\n\n        pos_embed = torch.randn(max_length, pos_embed_dim)\n        self.pos_embed = nn.Parameter(pos_embed)\n\n        self.cls_embed = nn.Parameter(torch.zeros((1, pos_embed_dim)))\n        self.x_embed = nn.Sequential(\n            nn.Linear(num_point * 3, pos_embed_dim, bias=False),\n        )\n\n        self.encoder = nn.LSTM(\n            input_size=pos_embed_dim,\n            hidden_size=lstm_hidden_size,\n            num_layers=lstm_num_layers,\n            batch_first=True,\n            bidirectional=False\n        )\n        self.fc = nn.Linear(lstm_hidden_size, num_class)\n\n    def forward(self, batch):\n        \n        #Keeping this part in since it seems nessesary.\n        length = [len(x) for x in batch['xyz']]\n        xyz = batch['xyz']\n\n        x, x_mask = pack_seq(xyz)\n        B,L,_ = x.shape\n        x = self.x_embed(x)\n        x = x + self.pos_embed[:L].unsqueeze(0)\n\n        x = torch.cat([\n            self.cls_embed.unsqueeze(0).repeat(B,1,1),\n            x\n        ], 1)\n\n        #Pack the sequence for the LSTM\n        x = nn.utils.rnn.pack_padded_sequence(x, length, batch_first=True, enforce_sorted=False) #Same idea as above. I want to see if the torch version is better.\n        output, (h_n, c_n) = self.encoder(x)\n\n        #Use the last output of the LSTM as the representation for regression\n        out = self.fc(h_n[-1])\n        #reg-loss should then be F.mse_loss(out, batch['reg'], reduction='mean')\n        return out\n","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:46:00.209953Z","iopub.execute_input":"2023-04-30T15:46:00.210506Z","iopub.status.idle":"2023-04-30T15:46:00.248438Z","shell.execute_reply.started":"2023-04-30T15:46:00.210462Z","shell.execute_reply":"2023-04-30T15:46:00.246319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_length = 96  #reduce this if gets out of memory error\n\nclass InputNet(nn.Module):\n    def __init__(self, ):\n        super().__init__()\n        self.max_length = max_length \n  \n    def forward(self, xyz):\n        xyz = xyz - xyz[~torch.isnan(xyz)].mean(0,keepdim=True) #noramlisation to common maen\n        xyz = xyz / xyz[~torch.isnan(xyz)].std(0, keepdim=True)\n\n        LIP = [\n            61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n            291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n            78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n            95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n        ]\n        #LHAND = np.arange(468, 489).tolist()\n        #RHAND = np.arange(522, 543).tolist()\n\n        lip = xyz[:, LIP]\n        lhand = xyz[:, 468:489]\n        rhand = xyz[:, 522:543]\n        xyz = torch.cat([  # (none, 82, 3)\n            lip,\n            lhand,\n            rhand,\n        ], 1)\n        xyz[torch.isnan(xyz)] = 0\n        x = xyz[:self.max_length]\n        return x\n\n\n#overwrite the model used in training ....\n\n# use fix dimension\nclass MultiHeadAttention(nn.Module):\n    def __init__(self,\n            embed_dim,\n            num_head,\n            batch_first,\n        ):\n        super().__init__()\n        self.mha = nn.MultiheadAttention(\n            embed_dim,\n            num_heads=num_head,\n            bias=True,\n            add_bias_kv=False,\n            kdim=None,\n            vdim=None,\n            dropout=0.0,\n            batch_first=batch_first,\n        )\n    #https://github.com/pytorch/text/blob/60907bf3394a97eb45056a237ca0d647a6e03216/torchtext/modules/multiheadattention.py#L5\n    def forward(self, x):\n        # out,_ = self.mha(x,x,x,need_weights=False)\n        # out,_ = F.multi_head_attention_forward(\n        #     x, x, x,\n        #     self.mha.embed_dim,\n        #     self.mha.num_heads,\n        #     self.mha.in_proj_weight,\n        #     self.mha.in_proj_bias,\n        #     self.mha.bias_k,\n        #     self.mha.bias_v,\n        #     self.mha.add_zero_attn,\n        #     0,#self.mha.dropout,\n        #     self.mha.out_proj.weight,\n        #     self.mha.out_proj.bias,\n        #     training=False,\n        #     key_padding_mask=None,\n        #     need_weights=False,\n        #     attn_mask=None,\n        #     average_attn_weights=False\n        # )\n \n        #qkv = F.linear(x, self.mha.in_proj_weight, self.mha.in_proj_bias)\n        #qkv = qkv.reshape(-1,3,1024)\n        #q,k,v = qkv[[0],0], qkv[:,1],  qkv[:,2]\n\n        q = F.linear(x[[0]], self.mha.in_proj_weight[:1024], self.mha.in_proj_bias[:1024]) #since we need only cls\n        k = F.linear(x, self.mha.in_proj_weight[1024:2048], self.mha.in_proj_bias[1024:2048])\n        v = F.linear(x, self.mha.in_proj_weight[2048:], self.mha.in_proj_bias[2048:]) \n        q = q.reshape(-1, 8, 128).permute(1, 0, 2)\n        k = k.reshape(-1, 8, 128).permute(1, 2, 0)\n        v = v.reshape(-1, 8, 128).permute(1, 0, 2)\n        dot  = torch.matmul(q, k) * (1/128**0.5) # H L L\n        attn = F.softmax(dot, -1)  #   L L\n        out  = torch.matmul(attn, v)  #   L H dim\n        out  = out.permute(1, 0, 2).reshape(-1, 1024)\n        out  = F.linear(out, self.mha.out_proj.weight, self.mha.out_proj.bias)  \n        return out\n\n# remove mask\nclass TransformerBlock(nn.Module):\n    def __init__(self,\n        embed_dim,\n        num_head,\n        out_dim,\n        batch_first=True,\n    ):\n        super().__init__()\n        self.attn  = MultiHeadAttention(embed_dim, num_head,batch_first)\n        self.ffn   = FeedForward(embed_dim, out_dim)\n        self.norm1 = nn.LayerNorm(embed_dim)\n        self.norm2 = nn.LayerNorm(out_dim)\n\n    def forward(self, x): \n        x = x + self.attn((self.norm1(x)))\n        x = x + self.ffn((self.norm2(x)))\n        return x\n\nclass SingleNet(nn.Module):\n\n    def __init__(self, num_class=num_class):\n        super().__init__()\n        self.num_block = 1\n        self.embed_dim = 1024\n        self.num_head  = 8\n        self.max_length = max_length\n        self.num_point = num_point\n\n        pos_embed = positional_encoding(max_length, self.embed_dim)\n        self.pos_embed = nn.Parameter(pos_embed)\n\n        self.cls_embed = nn.Parameter(torch.zeros((1, self.embed_dim)))\n        self.x_embed = nn.Sequential(\n            nn.Linear(num_point * 3, self.embed_dim, bias=False),\n        )\n\n        self.encoder = nn.ModuleList([\n            TransformerBlock(\n                self.embed_dim,\n                self.num_head,\n                self.embed_dim,\n                batch_first=False\n            ) for i in range(self.num_block)\n        ])\n        self.logit = nn.Linear(self.embed_dim, num_class)\n\n    def forward(self, xyz):\n        L = xyz.shape[0]\n        x_embed = self.x_embed(xyz.flatten(1)) \n        x = x_embed[:L] + self.pos_embed[:L]\n        x = torch.cat([\n            self.cls_embed,\n            x\n        ],0)\n        #x = x.unsqueeze(1)\n\n        #for block in self.encoder: x = block(x) #remove tflite loop\n        x = self.encoder[0](x)\n        cls = x[[0]]\n        logit = self.logit(cls)\n        return logit","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:46:09.152151Z","iopub.execute_input":"2023-04-30T15:46:09.152670Z","iopub.status.idle":"2023-04-30T15:46:09.186761Z","shell.execute_reply.started":"2023-04-30T15:46:09.152628Z","shell.execute_reply":"2023-04-30T15:46:09.184940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if 0:\n    \n    name='transformer-pool-2b' \n    input_onnx_file   = f'{fold_dir}/{name}.input.onnx'\n    single_onnx_file  = f'{fold_dir}/{name}.single.onnx' \n    input_tf_file    = f'{fold_dir}/input_tf'\n    single_tf_file   = f'{fold_dir}/single_tf'\n    tf_file     = f'{fold_dir}/tf'\n    tflite_file = f'{fold_dir}/{name}-{max_length}.tflite'\n\n    def run_convert_onnx(): \n        if 1:\n            torch.onnx.export(\n                input_net,\n                #torch.jit.script(input_net),\n                #torch.jit.trace(input_net, torch.zeros(100,num_landmark,3)),          # model being run \n                torch.zeros((100,num_landmark,3)), # model input (or a tuple for multiple inputs)\n                input_onnx_file,             # where to save the model (can be a file or file-like object)\n                export_params = True,        # store the trained parameter weights inside the model file\n                opset_version = 12,          # the ONNX version to export the model to\n                do_constant_folding=True,    # whether to execute constant folding for optimization \n                input_names =  ['inputs'],    # the model's input names\n                output_names = ['outputs'],   # the model's output names\n                dynamic_axes={\n                    'inputs': {0: 'length'},\n                    #'output': {0: 'length'},\n                },\n                #verbose = True,\n            )\n            torch.onnx.export(\n                single_net,         \n                #torch.jit.script(single_net),\n                #torch.jit.trace(single_net, torch.zeros(max_length,82,3)),           \n\n                torch.zeros((max_length,82,3)), \n                single_onnx_file,             \n                export_params = True,         \n                opset_version = 12, \n                do_constant_folding=True,      \n                input_names =  ['inputs'],     \n                output_names = ['outputs'],  \n                dynamic_axes={\n                    'inputs': {0: 'length'},\n                },\n                #verbose = True,\n            )\n            print('torch.onnx.export() passed !!')\n\n        if 1:\n            for f in [input_onnx_file, single_onnx_file]:\n                if f is None: continue\n                model = onnx.load(f)\n                onnx.checker.check_model(model)\n                model_simple, check = onnxsim.simplify(model)\n                onnx.save(model_simple, f)\n            print('onnx simplify() passed !!')\n\n\n    def run_convert_tflite():\n        if 1:\n            tf_rep = prepare(onnx.load(input_onnx_file))\n            tf_rep.export_graph(input_tf_file) \n            tf_rep = prepare(onnx.load(single_onnx_file))\n            tf_rep.export_graph(single_tf_file) \n            print('tf_rep.export_graph() passed !!')\n\n        if 1:\n            class TFModel(tf.Module):\n                def __init__(self):\n                    super(TFModel, self).__init__()\n                    self.input  = tf.saved_model.load(input_tf_file)\n                    self.single = tf.saved_model.load(single_tf_file)\n                    self.input.trainable = False\n                    self.single.trainable = False\n\n                @tf.function(input_signature=[\n                    tf.TensorSpec(shape=[None, 543, 3], dtype=tf.float32, name='inputs')\n                ])\n                def call(self, input):\n                    y = {}\n                    x = self.input(**{'inputs': input})['outputs']\n                    y['outputs'] = self.single(**{'inputs': x})['outputs'][0]\n                    return y\n\n            tfmodel = TFModel()\n            tf.saved_model.save(tfmodel, tf_file, signatures={'serving_default': tfmodel.call})\n            print('tf.saved_model() passed !!')\n\n        if 1:\n            converter = tf.lite.TFLiteConverter.from_saved_model(tf_file)\n            # converter.target_spec.supported_ops = [\n            #     tf.lite.OpsSet.TFLITE_BUILTINS,  # enable TensorFlow Lite ops.\n            #     tf.lite.OpsSet.SELECT_TF_OPS  # enable TensorFlow ops.\n            # ]\n            # converter.optimizations = [tf.lite.Optimize.DEFAULT]\n            #converter.allow_custom_ops = True\n            #converter.experimental_new_converter = True \n            tf_lite_model = converter.convert()\n            with open(tflite_file, 'wb') as f:\n                f.write(tf_lite_model)\n            print('tflite convert() passed !!')\n \n    run_convert_onnx()\n    run_convert_tflite()","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:46:12.824395Z","iopub.execute_input":"2023-04-30T15:46:12.825039Z","iopub.status.idle":"2023-04-30T15:46:12.850667Z","shell.execute_reply.started":"2023-04-30T15:46:12.824984Z","shell.execute_reply":"2023-04-30T15:46:12.849369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tflite_file = '/kaggle/input/asl-demo/transformer-pool-2b.tflite'   #max_length =180\ntflite_file = '/kaggle/input/asl-demo/transformer-pool-2b-96.tflite' #max_length =96 \ntflite_file = '/kaggle/input/asl-demo/transformer-pool-2c-512-80-fixed-int8.tflite'\n\ntflite_file = '/kaggle/input/asl-demo/transfomer-60-256-lip-hand-my-part-3a-int8.tflite'\nmode = 'submit' #debug #submit\n","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:46:15.722734Z","iopub.execute_input":"2023-04-30T15:46:15.723214Z","iopub.status.idle":"2023-04-30T15:46:15.729516Z","shell.execute_reply.started":"2023-04-30T15:46:15.723175Z","shell.execute_reply":"2023-04-30T15:46:15.728462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport shutil\nfrom datetime import datetime\nfrom timeit import default_timer as timer\n\n\nif mode in ['debug']:  \n    try:\n        import tflite_runtime\n    except:\n        !pip install tflite-runtime\n\n    import tflite_runtime.interpreter as tflite   \n    import tflite_runtime\n    print(tflite_runtime.__version__)\n    #'2.11.0'\n    \n    #import tensorflow as tf\n    #print(tf.__version__)\n    # 2.11.0\n\nprint('import ok')\n'''\nYour model must also require less than 40 MB in memory and \nperform inference with less than 100 milliseconds of latency per video. \nExpect to see approximately 40,000 videos in the test set. \nWe allow an additional 10 minute buffer for loading the data and miscellaneous overhead.\n\n'''\ndef time_to_str(t, mode='min'):\n    if mode=='min':\n        t  = int(t)/60\n        hr = t//60\n        min = t%60\n        return '%2d hr %02d min'%(hr,min)\n\n    elif mode=='sec':\n        t   = int(t)\n        min = t//60\n        sec = t%60\n        return '%2d min %02d sec'%(min,sec)\n\n    else:\n        raise NotImplementedError\n\n        \nROWS_PER_FRAME = 543\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)\n\nif mode in ['debug']: \n \n    interpreter = tflite.Interpreter(tflite_file)\n    prediction_fn = interpreter.get_signature_runner('serving_default')\n\n    valid_df = pd.read_csv('/kaggle/input/asl-demo/train_prepared.csv') \n    valid_df = valid_df[valid_df.fold==2].reset_index(drop=True)\n    valid_df = valid_df[:4_000]\n    valid_num = len(valid_df)\n    valid = {\n        'sign':[],\n    }\n\n    start_timer = timer()\n    for t, d in valid_df.iterrows():\n\n        pq_file = f'/kaggle/input/asl-signs/{d.path}'\n        #print(pq_file)\n        xyz = load_relevant_data_subset(pq_file)\n\n        output = prediction_fn(inputs=xyz)\n        p = output['outputs'].reshape(-1)\n\n        valid['sign'].append(p)\n\n        #---\n        if t%100==0:\n            time_taken = timer() - start_timer\n            print('\\r %8d / %d  %s'%(t,valid_num,time_to_str(time_taken,'sec')),end='',flush=True)\n\n    print('\\n')\n\n\n    truth = valid_df.label.values\n    sign  = np.stack(valid['sign'])\n    predict = np.argsort(-sign, -1)\n    correct = predict==truth.reshape(valid_num,1)\n    topk = correct.cumsum(-1).mean(0)[:5]\n\n\n    print(f'time_taken = {time_to_str(time_taken,\"sec\")}')\n    print(f'time_taken for LB = {time_taken*1000/valid_num:05f} msec\\n')\n    for i in range(5):\n        print(f'topk[{i}] = {topk[i]}')  \n    print('----- end -----\\n')\n\n\n\n\nshutil.copyfile(tflite_file, 'model.tflite') \n!zip submission.zip  'model.tflite'\n!ls\n\nprint('tflite_file:', tflite_file)\nprint(f'submit ok')","metadata":{"execution":{"iopub.status.busy":"2023-04-30T15:46:19.459082Z","iopub.execute_input":"2023-04-30T15:46:19.459498Z","iopub.status.idle":"2023-04-30T15:46:21.895241Z","shell.execute_reply.started":"2023-04-30T15:46:19.459464Z","shell.execute_reply":"2023-04-30T15:46:21.893540Z"},"trusted":true},"execution_count":null,"outputs":[]}]}