{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Ensemble of two models\nUsed 2 notebooks:\n* https://www.kaggle.com/code/jvthunder/lstm-baseline-for-starters-sign-language (with my own train  - LB = 0.6);\n* https://www.kaggle.com/code/roberthatch/gislr-lb-0-63-on-the-shoulders (LB = 0.63)\n","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, optimizers","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:56:26.010036Z","iopub.execute_input":"2023-03-20T14:56:26.010667Z","iopub.status.idle":"2023-03-20T14:56:26.016457Z","shell.execute_reply.started":"2023-03-20T14:56:26.010631Z","shell.execute_reply":"2023-03-20T14:56:26.015351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load models","metadata":{}},{"cell_type":"code","source":"asl_model = keras.models.load_model('/kaggle/input/gislr-tf-on-the-shoulders-s/models/asl_model')\nltsm_model = keras.models.load_model('/kaggle/input/sign-language-classification-2idat/lstm.h5')","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:56:26.018593Z","iopub.execute_input":"2023-03-20T14:56:26.019240Z","iopub.status.idle":"2023-03-20T14:56:32.018220Z","shell.execute_reply.started":"2023-03-20T14:56:26.019202Z","shell.execute_reply":"2023-03-20T14:56:32.017169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"asl_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:56:32.019728Z","iopub.execute_input":"2023-03-20T14:56:32.020144Z","iopub.status.idle":"2023-03-20T14:56:32.054455Z","shell.execute_reply.started":"2023-03-20T14:56:32.020085Z","shell.execute_reply":"2023-03-20T14:56:32.053729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ltsm_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:56:32.056716Z","iopub.execute_input":"2023-03-20T14:56:32.057040Z","iopub.status.idle":"2023-03-20T14:56:32.089164Z","shell.execute_reply.started":"2023-03-20T14:56:32.057004Z","shell.execute_reply":"2023-03-20T14:56:32.088397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Params and utils","metadata":{}},{"cell_type":"code","source":"ROWS_PER_FRAME = 543\nLANDMARK_IDX = [0,9,11,13,14,17,117,118,119,199,346,347,348] + list(range(468,543))","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:56:32.090058Z","iopub.execute_input":"2023-03-20T14:56:32.090384Z","iopub.status.idle":"2023-03-20T14:56:32.095509Z","shell.execute_reply.started":"2023-03-20T14:56:32.090350Z","shell.execute_reply":"2023-03-20T14:56:32.094768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DROP_Z = False\n\nNUM_FRAMES = 15\nSEGMENTS = 3\n\nLEFT_HAND_OFFSET = 468\nPOSE_OFFSET = LEFT_HAND_OFFSET+21\nRIGHT_HAND_OFFSET = POSE_OFFSET+33\n\n## average over the entire face, and the entire 'pose'\naveraging_sets = [[0, 468], [POSE_OFFSET, 33]]\n\nlip_landmarks = [61, 185, 40, 39, 37,  0, 267, 269, 270, 409,\n                 291,146, 91,181, 84, 17, 314, 405, 321, 375, \n                 78, 191, 80, 81, 82, 13, 312, 311, 310, 415, \n                 95, 88, 178, 87, 14,317, 402, 318, 324, 308]\nleft_hand_landmarks = list(range(LEFT_HAND_OFFSET, LEFT_HAND_OFFSET+21))\nright_hand_landmarks = list(range(RIGHT_HAND_OFFSET, RIGHT_HAND_OFFSET+21))\n\npoint_landmarks = [item for sublist in [lip_landmarks, left_hand_landmarks, right_hand_landmarks] for item in sublist]\n\nLANDMARKS = len(point_landmarks) + len(averaging_sets)\nprint(LANDMARKS)\nif DROP_Z:\n    INPUT_SHAPE = (NUM_FRAMES,LANDMARKS*2)\nelse:\n    INPUT_SHAPE = (NUM_FRAMES,LANDMARKS*3)\n\nFLAT_INPUT_SHAPE = (INPUT_SHAPE[0] + 2 * (SEGMENTS + 1)) * INPUT_SHAPE[1]","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:56:32.096769Z","iopub.execute_input":"2023-03-20T14:56:32.097155Z","iopub.status.idle":"2023-03-20T14:56:32.117701Z","shell.execute_reply.started":"2023-03-20T14:56:32.097104Z","shell.execute_reply":"2023-03-20T14:56:32.116915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tf_nan_mean(x, axis=0):\n    return tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), x), axis=axis) / tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), tf.ones_like(x)), axis=axis)\n\ndef tf_nan_std(x, axis=0):\n    d = x - tf_nan_mean(x, axis=axis)\n    return tf.math.sqrt(tf_nan_mean(d * d, axis=axis))\n\ndef flatten_means_and_stds(x, axis=0):\n    # Get means and stds\n    x_mean = tf_nan_mean(x, axis=0)\n    x_std  = tf_nan_std(x,  axis=0)\n\n    x_out = tf.concat([x_mean, x_std], axis=0)\n    x_out = tf.reshape(x_out, (1, INPUT_SHAPE[1]*2))\n    x_out = tf.where(tf.math.is_finite(x_out), x_out, tf.zeros_like(x_out))\n    return x_out\n\nclass FeatureGen(tf.keras.layers.Layer):\n    def __init__(self):\n        super(FeatureGen, self).__init__()\n    \n    def call(self, x_in):\n        if DROP_Z:\n            x_in = x_in[:, :, 0:2]\n        x_list = [tf.expand_dims(tf_nan_mean(x_in[:, av_set[0]:av_set[0]+av_set[1], :], axis=1), axis=1) for av_set in averaging_sets]\n        x_list.append(tf.gather(x_in, point_landmarks, axis=1))\n        x = tf.concat(x_list, 1)\n\n        x_padded = x\n        for i in range(SEGMENTS):\n            p0 = tf.where( ((tf.shape(x_padded)[0] % SEGMENTS) > 0) & ((i % 2) != 0) , 1, 0)\n            p1 = tf.where( ((tf.shape(x_padded)[0] % SEGMENTS) > 0) & ((i % 2) == 0) , 1, 0)\n            paddings = [[p0, p1], [0, 0], [0, 0]]\n            x_padded = tf.pad(x_padded, paddings, mode=\"SYMMETRIC\")\n        x_list = tf.split(x_padded, SEGMENTS)\n        x_list = [flatten_means_and_stds(_x, axis=0) for _x in x_list]\n\n        x_list.append(flatten_means_and_stds(x, axis=0))\n        \n        ## Resize only dimension 0. Resize can't handle nan, so replace nan with that dimension's avg value to reduce impact.\n        x = tf.image.resize(tf.where(tf.math.is_finite(x), x, tf_nan_mean(x, axis=0)), [NUM_FRAMES, LANDMARKS])\n        x = tf.reshape(x, (1, INPUT_SHAPE[0]*INPUT_SHAPE[1]))\n        x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n        x_list.append(x)\n        x = tf.concat(x_list, axis=1)\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:56:32.119043Z","iopub.execute_input":"2023-03-20T14:56:32.119381Z","iopub.status.idle":"2023-03-20T14:56:32.146223Z","shell.execute_reply.started":"2023-03-20T14:56:32.119347Z","shell.execute_reply":"2023-03-20T14:56:32.145132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Make ensemble ","metadata":{}},{"cell_type":"code","source":"def get_inference_model(model,asl_model):\n    inputs = tf.keras.Input(shape=(ROWS_PER_FRAME,3), name=\"inputs\")\n    \n    # drop most of the face mesh\n    x = tf.gather(inputs, LANDMARK_IDX, axis=1)\n\n    # fill nan\n    x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n\n    # flatten landmark xyz coordinates ()\n    x = tf.concat([x[...,i] for i in range(3)], -1)\n\n    x = tf.expand_dims(x,0)\n    \n    # call trained model\n    out = model(x)\n    \n    #model 2\n    prep_inputs = FeatureGen()\n    x2 = prep_inputs(tf.cast(inputs, dtype=tf.float32))\n    out2 = asl_model(x2)\n    \n    #out3 = tf.keras.layers.Average()([out,out2])\n    out3 = tf.keras.layers.Multiply()([out,out2])\n    \n    # explicitly name the final (identity) layer for the submission format\n    out = layers.Activation(\"linear\", name=\"outputs\")(out3)\n    \n    inference_model = tf.keras.Model(inputs=inputs, outputs=out)\n    inference_model.compile(loss=\"sparse_categorical_crossentropy\",\n                            metrics=\"accuracy\")\n    return inference_model","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:56:32.147813Z","iopub.execute_input":"2023-03-20T14:56:32.148547Z","iopub.status.idle":"2023-03-20T14:56:32.163184Z","shell.execute_reply.started":"2023-03-20T14:56:32.148497Z","shell.execute_reply":"2023-03-20T14:56:32.161878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_model = get_inference_model(ltsm_model,asl_model)\ntf.keras.utils.plot_model(inference_model)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:56:32.164925Z","iopub.execute_input":"2023-03-20T14:56:32.165666Z","iopub.status.idle":"2023-03-20T14:56:33.450631Z","shell.execute_reply.started":"2023-03-20T14:56:32.165631Z","shell.execute_reply":"2023-03-20T14:56:33.449362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_model.summary(expand_nested=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:56:33.453823Z","iopub.execute_input":"2023-03-20T14:56:33.454234Z","iopub.status.idle":"2023-03-20T14:56:33.542910Z","shell.execute_reply.started":"2023-03-20T14:56:33.454098Z","shell.execute_reply":"2023-03-20T14:56:33.542161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Convert model","metadata":{}},{"cell_type":"code","source":"converter = tf.lite.TFLiteConverter.from_keras_model(inference_model)\ntflite_model = converter.convert()\nmodel_path = \"model.tflite\"\n\nwith open(model_path, 'wb') as f:\n    f.write(tflite_model)\n!zip submission.zip $model_path","metadata":{"execution":{"iopub.status.busy":"2023-03-20T14:56:33.543898Z","iopub.execute_input":"2023-03-20T14:56:33.544256Z","iopub.status.idle":"2023-03-20T14:56:54.889032Z","shell.execute_reply.started":"2023-03-20T14:56:33.544221Z","shell.execute_reply":"2023-03-20T14:56:54.887557Z"},"trusted":true},"execution_count":null,"outputs":[]}]}