{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Isolated Sign Language Recognition : dive in \n\nThis notebook is the continuation of my previous notebook [here](https://www.kaggle.com/code/dorianmb/isolated-sign-language-recognition-quick-start?scriptVersionId=121909882)\n\n","metadata":{}},{"cell_type":"markdown","source":"# Credits\n\nNotebook that help me : \n- [NGHI HUYNH's notebook](https://www.kaggle.com/code/nghihuynh/gislr-eda-feature-processing)\n- [ROBERT HATCH' notbook](https://www.kaggle.com/code/roberthatch/gislr-lb-0-63-on-the-shoulders/notebook#PREPROCESSING)\n- [JVTHUNDER's notebook](https://www.kaggle.com/code/jvthunder/lstm-baseline-for-starters-sign-language)\n","metadata":{}},{"cell_type":"markdown","source":"# As newbie on kaggle, I am open to any suggestion for improvement.","metadata":{}},{"cell_type":"markdown","source":"# Import","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers\nfrom sklearn.model_selection import train_test_split\n\nimport os\nimport random\nimport json","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-01T12:59:09.506960Z","iopub.execute_input":"2023-04-01T12:59:09.507418Z","iopub.status.idle":"2023-04-01T12:59:19.456842Z","shell.execute_reply.started":"2023-04-01T12:59:09.507368Z","shell.execute_reply":"2023-04-01T12:59:19.455684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Install nb_black for autoformatting\n!pip install nb_black --quiet\n%load_ext lab_black","metadata":{"execution":{"iopub.status.busy":"2023-03-25T13:16:05.130957Z","iopub.execute_input":"2023-03-25T13:16:05.131994Z","iopub.status.idle":"2023-03-25T13:16:20.242363Z","shell.execute_reply.started":"2023-03-25T13:16:05.131954Z","shell.execute_reply":"2023-03-25T13:16:20.241128Z"}}},{"cell_type":"code","source":"print(\"tensorflow version : \", tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:19.459192Z","iopub.execute_input":"2023-04-01T12:59:19.460025Z","iopub.status.idle":"2023-04-01T12:59:19.466258Z","shell.execute_reply.started":"2023-04-01T12:59:19.459985Z","shell.execute_reply":"2023-04-01T12:59:19.464855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Set-up & Reproducible","metadata":{}},{"cell_type":"code","source":"# Constante\nSEED = 42\nROWS_PER_FRAME = 543\ndata_dir = \"/kaggle/input/asl-signs\"\nlandmark_fimes_dir = \"/kaggle/input/asl-signs/train_landmark_files\"","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:19.467999Z","iopub.execute_input":"2023-04-01T12:59:19.468366Z","iopub.status.idle":"2023-04-01T12:59:19.476643Z","shell.execute_reply.started":"2023-04-01T12:59:19.468316Z","shell.execute_reply":"2023-04-01T12:59:19.475494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seed_it_all(seed=SEED):\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\n\nseed_it_all()  # Reproducible","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:19.478410Z","iopub.execute_input":"2023-04-01T12:59:19.478721Z","iopub.status.idle":"2023-04-01T12:59:19.489864Z","shell.execute_reply.started":"2023-04-01T12:59:19.478691Z","shell.execute_reply":"2023-04-01T12:59:19.488696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_relevant_data_subset(pq_path):\n    data_columns = [\"x\", \"y\", \"z\"]\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:19.492916Z","iopub.execute_input":"2023-04-01T12:59:19.493317Z","iopub.status.idle":"2023-04-01T12:59:19.504060Z","shell.execute_reply.started":"2023-04-01T12:59:19.493262Z","shell.execute_reply":"2023-04-01T12:59:19.503028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_json(path):\n    with open(path, \"r\") as file:\n        json_data = json.load(file)\n    return json_data","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:19.505226Z","iopub.execute_input":"2023-04-01T12:59:19.506784Z","iopub.status.idle":"2023-04-01T12:59:19.516197Z","shell.execute_reply.started":"2023-04-01T12:59:19.506729Z","shell.execute_reply":"2023-04-01T12:59:19.515199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load data","metadata":{}},{"cell_type":"code","source":"path_train_df = pd.read_csv(data_dir + \"/train.csv\")\npath_train_df[\"path\"] = data_dir + \"/\" + path_train_df[\"path\"]\ndisplay(path_train_df.head(2)), len(path_train_df)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:19.517385Z","iopub.execute_input":"2023-04-01T12:59:19.517804Z","iopub.status.idle":"2023-04-01T12:59:19.789325Z","shell.execute_reply.started":"2023-04-01T12:59:19.517756Z","shell.execute_reply":"2023-04-01T12:59:19.788521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s2p_map = read_json(os.path.join(data_dir, \"sign_to_prediction_index_map.json\"))\np2s_map = {v: k for k, v in s2p_map.items()}\n\nencoder = lambda x: s2p_map.get(x)\ndecoder = lambda x: p2s_map.get(x)\n\npath_train_df[\"label\"] = path_train_df[\"sign\"].map(encoder)\n\ndisplay(path_train_df.head(2)), path_train_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:19.790591Z","iopub.execute_input":"2023-04-01T12:59:19.791534Z","iopub.status.idle":"2023-04-01T12:59:19.858602Z","shell.execute_reply.started":"2023-04-01T12:59:19.791495Z","shell.execute_reply":"2023-04-01T12:59:19.857492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_parquet(path_train_df.path[0])\nsample.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:19.860159Z","iopub.execute_input":"2023-04-01T12:59:19.860929Z","iopub.status.idle":"2023-04-01T12:59:19.993817Z","shell.execute_reply.started":"2023-04-01T12:59:19.860878Z","shell.execute_reply":"2023-04-01T12:59:19.992930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**need to select usefull Landmark index**","metadata":{}},{"cell_type":"markdown","source":"# Quick EDA and Preprocessing","metadata":{}},{"cell_type":"markdown","source":"## Distribution of number of frame per sequence over the dataset","metadata":{}},{"cell_type":"code","source":"# let's check the distribution of n_frame over the dataset\ndistribution_lenght = int(len(path_train_df) / 100)\nframes = np.zeros(distribution_lenght)\nfor index, row in tqdm(path_train_df.iterrows(), total=distribution_lenght):\n    if index > distribution_lenght - 1:\n        break\n    x = load_relevant_data_subset(row.path)\n    frames[index] = x.shape[0]","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:19.994876Z","iopub.execute_input":"2023-04-01T12:59:19.995771Z","iopub.status.idle":"2023-04-01T12:59:35.716535Z","shell.execute_reply.started":"2023-04-01T12:59:19.995733Z","shell.execute_reply":"2023-04-01T12:59:35.715701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"----------------------------------------------------\")\nprint(\"minimum frames in sequence : \", frames.min())\nprint(\"maximum frames in sequence : \", frames.max())\nprint(f\"mean of frames in the first {distribution_lenght} sequences\", frames.mean())\nprint(f\"median of frames in the first {distribution_lenght} sequences\", np.median(frames))\nprint(\"----------------------------------------------------\")","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:35.717819Z","iopub.execute_input":"2023-04-01T12:59:35.718804Z","iopub.status.idle":"2023-04-01T12:59:35.726959Z","shell.execute_reply.started":"2023-04-01T12:59:35.718764Z","shell.execute_reply":"2023-04-01T12:59:35.725498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10, 5))\nhist = plt.hist(frames, bins=100)\n\nmedian = np.median(frames)\nplt.plot(\n    [median, median],\n    [0, hist[0].max()],\n    \"--\",\n    c=\"red\",\n    linewidth=2,\n    label=f\"Median={median:.0f} frames\",\n)\n\nplt.title(\"Distribution of number of frame per sequence over the dataset\")\nplt.xlabel(\"n-frames in one sequence\")\nplt.ylabel(\"Number of sequence with n-frames\")\n\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:35.729137Z","iopub.execute_input":"2023-04-01T12:59:35.729512Z","iopub.status.idle":"2023-04-01T12:59:36.209660Z","shell.execute_reply.started":"2023-04-01T12:59:35.729475Z","shell.execute_reply":"2023-04-01T12:59:36.208413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"int(np.percentile(frames, 25))","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:36.211292Z","iopub.execute_input":"2023-04-01T12:59:36.211868Z","iopub.status.idle":"2023-04-01T12:59:36.220826Z","shell.execute_reply.started":"2023-04-01T12:59:36.211816Z","shell.execute_reply":"2023-04-01T12:59:36.219586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Herlper function","metadata":{}},{"cell_type":"code","source":"def tf_nan_mean(x, axis=0):\n    x_zero_insted_of_nan = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n    x_zero_for_nan_else_one = tf.where(\n        tf.math.is_nan(x), tf.zeros_like(x), tf.ones_like(x)\n    )\n    x_out = tf.reduce_sum(x_zero_insted_of_nan, axis=axis) / tf.reduce_sum(\n        x_zero_for_nan_else_one, axis=axis\n    )\n    return tf.where(tf.math.is_finite(x_out), x_out, tf.zeros_like(x_out))\n\n\ndef tf_nan_std(x, axis=0):\n    d = x - tf_nan_mean(x, axis=axis)\n    return tf.math.sqrt(tf_nan_mean(d * d, axis=axis))","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:36.227087Z","iopub.execute_input":"2023-04-01T12:59:36.227957Z","iopub.status.idle":"2023-04-01T12:59:36.237507Z","shell.execute_reply.started":"2023-04-01T12:59:36.227915Z","shell.execute_reply":"2023-04-01T12:59:36.235958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"DROP_Z = False\n\n# Drop most of the face landmarks to reduce the dimensionality\nLANDMARK = [0, 9, 11, 13, 14, 17, 117, 118, 119, 199, 346, 347, 348] + list(\n    range(468, 543)\n)\n\nLENGHT_LANDMARK = len(LANDMARK)\nN_DATA = len([\"x\", \"y\"]) if DROP_Z else len([\"x\", \"y\", \"z\"])\nFIXED_FRAME = int(np.median(frames))\nSHAPE = [FIXED_FRAME, LENGHT_LANDMARK, N_DATA]\n\nprint(\"----------------------------------------------------\")\nprint(\"Drop Z in data column :\", DROP_Z, end=\"\\n\\n\")\nprint(\"Fixed Frame (shape[0]) =\", FIXED_FRAME, end=\"\\n\\n\")\nprint(\"Shape =\", SHAPE, end=\"\\n\\n\")\nprint(\"----------------------------------------------------\")","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:36.239589Z","iopub.execute_input":"2023-04-01T12:59:36.239989Z","iopub.status.idle":"2023-04-01T12:59:36.257771Z","shell.execute_reply.started":"2023-04-01T12:59:36.239951Z","shell.execute_reply":"2023-04-01T12:59:36.256376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Generator","metadata":{}},{"cell_type":"code","source":"class FeatureGen(tf.keras.layers.Layer):\n    def __init__(self):\n        super().__init__()\n\n    def call(self, x):\n        if x.shape[0] is None :\n            n_frames = FIXED_FRAME\n        else :\n            n_frames = x.shape[0]\n\n        # Drop \"z\" column\n        if DROP_Z:\n            x = x[:, :, 0:2]\n\n        # NaN values become 0\n        x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n\n        # Landmarks reduction\n        # Select only the usefull landmark\n        x = tf.gather(\n            x,\n            indices=LANDMARK,\n            axis=1,\n        )\n        \n        if FIXED_FRAME > n_frames:\n            outputs = tf.image.resize(x, size=[SHAPE[0], SHAPE[1]], method=\"bilinear\")\n        else:\n            outputs = tf.image.resize(x, size=[SHAPE[0], SHAPE[1]], method=\"nearest\")\n\n        return outputs\n\n\nfeature_converter = FeatureGen()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:28:35.362391Z","iopub.execute_input":"2023-04-01T13:28:35.362796Z","iopub.status.idle":"2023-04-01T13:28:35.373037Z","shell.execute_reply.started":"2023-04-01T13:28:35.362738Z","shell.execute_reply":"2023-04-01T13:28:35.372118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Some test\nsample = load_relevant_data_subset(path_train_df.path[1])\nprepocesse_sample = feature_converter(sample)\nprepocesse_sample.shape, sample.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:36.284720Z","iopub.execute_input":"2023-04-01T12:59:36.285083Z","iopub.status.idle":"2023-04-01T12:59:36.440393Z","shell.execute_reply.started":"2023-04-01T12:59:36.285047Z","shell.execute_reply":"2023-04-01T12:59:36.438805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Making data","metadata":{}},{"cell_type":"code","source":"TOTAL_DATA_LENGHT = len(path_train_df)\nDATA_LENGHT_EXPERIMENT = int(len(path_train_df) / 100)\n\nprint(\"----------------------------------------------------\")\nprint(\"Lenght of data for modeling :\", DATA_LENGHT_EXPERIMENT)\nprint(f\"Percentage of total data {DATA_LENGHT_EXPERIMENT/TOTAL_DATA_LENGHT*100:.1f}%\")\nprint(\"----------------------------------------------------\")","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:36.442200Z","iopub.execute_input":"2023-04-01T12:59:36.442878Z","iopub.status.idle":"2023-04-01T12:59:36.450688Z","shell.execute_reply.started":"2023-04-01T12:59:36.442837Z","shell.execute_reply":"2023-04-01T12:59:36.449201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_row(row):\n    x = load_relevant_data_subset(row.path)\n    x = feature_converter(x)\n    return x, row.label\n\n\ndef convert_and_save_data(data_lenght=DATA_LENGHT_EXPERIMENT):\n    np_features = np.zeros([data_lenght] + SHAPE)\n    np_labels = np.zeros(data_lenght)\n    \n    print(\"----------------------------------------------------\")\n    print(f\"Total data to processe : {data_lenght}\")\n    print(f\"Percentage of total data {data_lenght/TOTAL_DATA_LENGHT*100:.2f}%\")\n    print(\"----------------------------------------------------\")\n    \n    for index, row in tqdm(path_train_df.iterrows(), total=data_lenght):\n        if index > data_lenght - 1:\n            break\n\n        if index % (DATA_LENGHT_EXPERIMENT // 10) == 0:\n            print(f\"Data processed {index/data_lenght*100:.1f}%\")\n\n        data = load_relevant_data_subset(row.path)\n        feature, label = convert_row(row)\n        np_features[index] = feature\n        np_labels[index] = label\n\n    np.save(\"features.npy\", np_features)\n    np.save(\"labels.npy\", np_labels)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:36.452421Z","iopub.execute_input":"2023-04-01T12:59:36.453182Z","iopub.status.idle":"2023-04-01T12:59:36.463776Z","shell.execute_reply.started":"2023-04-01T12:59:36.453143Z","shell.execute_reply":"2023-04-01T12:59:36.462366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    features = np.load(\"/kaggle/working/features.npy\")\n    labels = np.load(\"/kaggle/working/labels.npy\")\n    print(\"Data Load successfully\")\nexcept:\n    print(\"Loading DATA has fail... \\nCreating DataSet\")\n    convert_and_save_data(DATA_LENGHT_EXPERIMENT)\n\n    features = np.load(\"features.npy\")\n    labels = np.load(\"labels.npy\")","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:36.465160Z","iopub.execute_input":"2023-04-01T12:59:36.465525Z","iopub.status.idle":"2023-04-01T12:59:49.198543Z","shell.execute_reply.started":"2023-04-01T12:59:36.465490Z","shell.execute_reply":"2023-04-01T12:59:49.197363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_size=0.2\nX_train, X_val, y_train, y_val = train_test_split(\n    features, labels, test_size=test_size, random_state=SEED\n)\n\ndel features,labels # delete, usefull with full data otherwise it fail (memorry issues)\n\nbuffer_size = int(DATA_LENGHT_EXPERIMENT / 10)\n\ntrain_data = tf.data.Dataset.from_tensor_slices((X_train, y_train))\ntrain_data = train_data.shuffle(buffer_size).batch(128, drop_remainder=True).prefetch(tf.data.AUTOTUNE)\n\nval_data = tf.data.Dataset.from_tensor_slices((X_val, y_val))\nval_data = val_data.batch(128, drop_remainder=True).prefetch(tf.data.AUTOTUNE)\n\nprint(\"----------------------------------------------------\")\nprint(\"X_train shape = \", X_train.shape)\nprint(\"X_val shape = \", X_val.shape)\nprint(\"y_train shape = \", y_train.shape)\nprint(\"y_val shape = \", y_val.shape)\nprint(\"----------------------------------------------------\")","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:49.200283Z","iopub.execute_input":"2023-04-01T12:59:49.200915Z","iopub.status.idle":"2023-04-01T12:59:49.314539Z","shell.execute_reply.started":"2023-04-01T12:59:49.200865Z","shell.execute_reply":"2023-04-01T12:59:49.313158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Quick test -> after training\nquick_test_idx = np.random.randint(0, len(y_val), size=(10,))  # See after training\nquick_test_X = np.take(X_val, quick_test_idx, axis=0)\nquick_test_y = np.take(y_val, quick_test_idx, axis=0)\ndel X_train, X_val, y_train, y_val","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:49.316267Z","iopub.execute_input":"2023-04-01T12:59:49.317343Z","iopub.status.idle":"2023-04-01T12:59:49.325707Z","shell.execute_reply.started":"2023-04-01T12:59:49.317282Z","shell.execute_reply":"2023-04-01T12:59:49.324573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Modeling","metadata":{}},{"cell_type":"code","source":"class DenseBlock(layers.Layer):\n    def __init__(self, units, drop):\n        super().__init__()\n        self.dense = layers.Dense(units)\n        self.norm = layers.LayerNormalization()\n        self.relu = layers.Activation(\"relu\")\n        self.drop = layers.Dropout(drop)\n        \n    def call(self, x):\n        x = self.dense(x)\n        x = self.norm(x)\n        x = self.relu(x)\n        x = self.drop(x)\n        return x\n\nclass ClassifierLSTM(layers.Layer):\n    def __init__(self, lstm_units, drop):\n        super().__init__()\n        self.dropout = layers.Dropout(drop)\n        self.pool2d = layers.AveragePooling2D(pool_size=(4, 1)) # Keep the same number of landmark\n        self.reshape = layers.Reshape((-1, 64))\n        self.lstm = layers.LSTM(units=lstm_units, return_sequences=True)\n        self.pool1d = layers.AveragePooling1D(pool_size=4)\n        self.flat = layers.Flatten()\n        self.outputs = layers.Dense(250, activation=\"softmax\", name=\"predictions\")\n    \n    def call(self, x): # (None, 23, 88, 64)\n        x = self.pool2d(x) # (None, 5, 88, 64)\n        x = self.reshape(x) # (None, 2024, 64)\n        x = self.lstm(x) # (None, 2024, 250)\n        x = self.dropout(x)\n        x = self.pool1d(x) # (None, 506, 250)\n        x = self.flat(x) # (None, 126_500)\n        outputs = self.outputs(x) # (None, 126 _00)\n        return outputs\n\n\nclass ClassifierConvLSTM1D(layers.Layer):\n    def __init__(self, lstm_units, drop):\n        super().__init__()\n        self.pool2d = layers.AveragePooling2D(pool_size=(6, 1)) # Keep the same number of landmark\n        self.conv_lstm1D = layers.ConvLSTM1D(filters=lstm_units, kernel_size=1) # RNN capable of learning long-term dependencies\n        self.dropout = layers.Dropout(drop)\n        self.flat = layers.Flatten()\n        self.outputs = layers.Dense(250, activation=\"softmax\", name=\"predictions\")\n    \n    def call(self, x): # (None, 23, 88, 64)\n        x = self.pool2d(x) # (None, 5, 88, 64)\n        x = self.conv_lstm1D(x) # (None, 88, 250)\n        x = self.dropout(x)\n        x = self.flat(x)\n        outputs = self.outputs(x)\n        return outputs","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:49.328979Z","iopub.execute_input":"2023-04-01T12:59:49.329479Z","iopub.status.idle":"2023-04-01T12:59:49.346422Z","shell.execute_reply.started":"2023-04-01T12:59:49.329442Z","shell.execute_reply":"2023-04-01T12:59:49.345276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model(\n    encoder_units=[128, 64],\n    drop=0.5,\n    lstm_units=250,\n    shape=SHAPE,\n    learning_rate=0.001,\n):\n    inputs = layers.Input(shape=shape)\n    x = inputs\n\n    for units in encoder_units:\n        x = DenseBlock(units, drop)(x)\n\n    outputs = ClassifierConvLSTM1D(lstm_units, drop)(x)\n    # ClassifierConvLSTM1D\n    # ClassifierLSTM\n\n    model = tf.keras.Model(inputs=inputs, outputs=outputs)\n\n    model.compile(\n        loss=\"sparse_categorical_crossentropy\",\n        optimizer=tf.keras.optimizers.Adam(learning_rate=learning_rate),\n        metrics=[\"accuracy\"],\n    )\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:24:17.138867Z","iopub.execute_input":"2023-04-01T13:24:17.139311Z","iopub.status.idle":"2023-04-01T13:24:17.148219Z","shell.execute_reply.started":"2023-04-01T13:24:17.139256Z","shell.execute_reply":"2023-04-01T13:24:17.146686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_callbacks():\n    return [\n        tf.keras.callbacks.EarlyStopping(\n            monitor=\"val_accuracy\", patience=10, restore_best_weights=True\n        ),\n        tf.keras.callbacks.ReduceLROnPlateau(\n            monitor=\"val_accuracy\", factor=0.5, patience=3\n        ),\n        tf.keras.callbacks.ModelCheckpoint(\n            \"./ASL_model\",\n            save_best_only=True,\n            restore_best_weights=True,\n            monitor=\"val_accuracy\",\n            mode=\"max\",\n            verbose=False,\n        ),\n    ]\n\n\ncb_list = get_callbacks()\n\nmodel = get_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:24:17.283487Z","iopub.execute_input":"2023-04-01T13:24:17.283942Z","iopub.status.idle":"2023-04-01T13:24:17.684242Z","shell.execute_reply.started":"2023-04-01T13:24:17.283898Z","shell.execute_reply":"2023-04-01T13:24:17.682767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.utils import plot_model\n\nplot_model(model, expand_nested=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:24:17.686267Z","iopub.execute_input":"2023-04-01T13:24:17.686641Z","iopub.status.idle":"2023-04-01T13:24:17.893049Z","shell.execute_reply.started":"2023-04-01T13:24:17.686605Z","shell.execute_reply":"2023-04-01T13:24:17.891429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"----------------------------------------------------\")\nprint(\"Guessing: \", 1 / 250)\nprint(\"Data per Class:\", DATA_LENGHT_EXPERIMENT*(1-test_size) / 250)\nprint(\"----------------------------------------------------\")","metadata":{"execution":{"iopub.status.busy":"2023-04-01T12:59:50.228563Z","iopub.execute_input":"2023-04-01T12:59:50.229101Z","iopub.status.idle":"2023-04-01T12:59:50.235773Z","shell.execute_reply.started":"2023-04-01T12:59:50.229062Z","shell.execute_reply":"2023-04-01T12:59:50.234623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fit","metadata":{"execution":{"iopub.status.busy":"2023-03-25T11:00:48.764672Z","iopub.execute_input":"2023-03-25T11:00:48.765295Z","iopub.status.idle":"2023-03-25T11:00:48.798007Z","shell.execute_reply.started":"2023-03-25T11:00:48.765217Z","shell.execute_reply":"2023-03-25T11:00:48.796143Z"}}},{"cell_type":"code","source":"%%time\nhistory = model.fit(\n    train_data,\n    validation_data=val_data,\n    epochs=1,\n    callbacks=cb_list\n)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:24:20.959128Z","iopub.execute_input":"2023-04-01T13:24:20.960216Z","iopub.status.idle":"2023-04-01T13:24:49.185645Z","shell.execute_reply.started":"2023-04-01T13:24:20.960127Z","shell.execute_reply":"2023-04-01T13:24:49.184396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.models.load_model(\"./ASL_model\")\nscore = model.evaluate(val_data)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:00:18.230158Z","iopub.execute_input":"2023-04-01T13:00:18.230531Z","iopub.status.idle":"2023-04-01T13:00:21.323760Z","shell.execute_reply.started":"2023-04-01T13:00:18.230495Z","shell.execute_reply":"2023-04-01T13:00:21.322518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model.predict(quick_test_X, verbose=False).argmax(axis=1)\n\nfor true_label_id, pred_label_id in zip(quick_test_y, predictions):\n    true_label = decoder(true_label_id)\n    pred_label = decoder(pred_label_id)\n    result = True if pred_label == true_label else False\n    print(\n        f\"Prediction on val label : {pred_label.upper():<10} => True label {true_label.upper():<10} => {int(result)}\"\n    )","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:00:21.325223Z","iopub.execute_input":"2023-04-01T13:00:21.325716Z","iopub.status.idle":"2023-04-01T13:00:21.726097Z","shell.execute_reply.started":"2023-04-01T13:00:21.325677Z","shell.execute_reply":"2023-04-01T13:00:21.725039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Benchmark on 10% data ConvLSTM1D\n\n`kernel_size=1`, 192 landmark, 13 frames, `lstm_units=100`, `units=[512, 256]` :\n- 2dense block => 24s 402ms/step - loss: 0.1679 - accuracy: 0.9951 - val_loss: 6.1105 - val_accuracy: 0.1402 - lr: 3.1250e-05\n- 2dense block without Z =>23s 388ms/step - loss: 0.1138 - accuracy: 0.9978 - val_loss: 5.6463 - val_accuracy: 0.1810 - lr: 6.2500e-05\n- 2dense block kernel_size=4 => 85s 1s/step - loss: 5.5149 - accuracy: 0.0060 - val_loss: 5.5236 - val_accuracy: 0.0032 - lr: 1.2500e-04 - best accuracy: 0.0069\n- 2dense block only \"bilinear\" => 27s 457ms/step - loss: 0.0523 - accuracy: 0.9996 - val_loss: 5.9966 - val_accuracy: 0.1852 - lr: 6.2500e-05\n\n---\n\n`kernel_size=1`, 192 landmark, 13 frames, `lstm_units=100`, `units=[512, 256, 128]` :\n- 3dense block only \"bilinear\" => 29s 477ms/step - loss: 0.0864 - accuracy: 0.9959 - val_loss: 6.2813 - val_accuracy: 0.1884 - lr: 3.1250e-05 - best accuracy: 0.1926\n- 3dense block only \"nearest\" => 28s 471ms/step - loss: 0.6903 - accuracy: 0.8809 - val_loss: 5.1938 - val_accuracy: 0.1825 - lr: 3.1250e-05 - best accuracy: 0.1857\n- 3dense block only \"bilinear\" => 25s 421ms/step - loss: 0.5771 - accuracy: 0.9135 - val_loss: 5.7056 - val_accuracy: 0.1608 - lr: 3.1250e-05 - best accuracy: 0.1608\n- 3dense block => 26s 440ms/step - loss: 0.0279 - accuracy: 0.9997 - val_loss: 6.7141 - val_accuracy: 0.1899 - lr: 1.5625e-05 - best accuracy: 0.1952\n- 3dense block + normalisation => 27s 441ms/step - loss: 0.0011 - accuracy: 1.0000 - val_loss: 11.0349 - val_accuracy: 0.0587 - lr: 1.2500e-04 - best accuracy: 0.0614\n\n---\n\n2dense block 23frames `units=[128, 64]`\n- 3dense block 45frames `units=[512, 256, 128]`=> 57s 953ms/step - loss: 0.1020 - accuracy: 0.9794 - val_loss: 6.4184 - val_accuracy: 0.1868 - lr: 6.2500e-05 - best accuracy: 0.1884\n- 3dense block 23frames `units=[512, 256, 128]`=> 33s 544ms/step - loss: 0.1231 - accuracy: 0.9790 - val_loss: 5.8464 - val_accuracy: 0.1989 - lr: 1.5625e-05 - best accuracy: 0.2042\n- 2dense block 23frames `units=[512, 256]` => 33s 544ms/step - loss: 0.0770 - accuracy: 0.9874 - val_loss: 6.0699 - val_accuracy: 0.1974 - lr: 3.1250e-05 - best accuracy: 0.2021\n- 2dense block 23frames `units=[256, 128]` => 24s 407ms/step - loss: 0.0993 - accuracy: 0.9807 - val_loss: 6.0558 - val_accuracy: 0.2238 - lr: 3.1250e-05 - best accuracy: 0.2286\n- 2dense block 23frames `units=[128, 64]` => 22s 362ms/step - loss: 0.4469 - accuracy: 0.8834 - val_loss: 5.4353 - val_accuracy: 0.2365 - lr: 6.2500e-05 - best accuracy: 0.2381\n- 2dense block 23frames `units=[64, 32]` => 20s 333ms/step - loss: 0.3328 - accuracy: 0.9096 - val_loss: 6.2471 - val_accuracy: 0.2143 - lr: 3.1250e-05 - best accuracy: 0.2196\n- 2dense block 23frames without_Z `units=[128, 64]` => 22s 362ms/step - loss: 0.1107 - accuracy: 0.9746 - val_loss: 6.9636 - val_accuracy: 0.2069 - lr: 1.5625e-05 - best accuracy: 0.2074\n\n---\n2dense block 23frames, `units=[128, 64]`, `units=[128, 64]`\n- 7s 112ms/step - loss: 0.1199 - accuracy: 0.9707 - val_loss: 6.8240 - val_accuracy: 0.2260 - lr: 6.2500e-05","metadata":{}},{"cell_type":"markdown","source":"## Benchmark on 100% of data\n\n- 2dense block 23frames `units=[128, 64]` => 219s 370ms/step - loss: 0.3757 - accuracy: 0.8876 - val_loss: 2.3673 - val_accuracy: 0.5853 - lr: 1.9531e-06","metadata":{}},{"cell_type":"markdown","source":"## Benchmark on 10% data LSTM\n\n- 9s 149ms/step - loss: 0.0156 - accuracy: 0.9977 - val_loss: 9.6546 - val_accuracy: 0.1702 - lr: 6.2500e-05 - best accuracy: 0.1769","metadata":{}},{"cell_type":"markdown","source":"# Visualize History","metadata":{}},{"cell_type":"code","source":"def plot_history(history, zoom=0):\n    df = pd.DataFrame(history.history)\n    n = len(df.columns)\n\n    row = n // 2\n    col = n // 2 + n % 2\n\n    plt.figure(figsize=(5 * (col + 1) + zoom, 5 * row + zoom))\n    for i, column in enumerate(df.columns):\n        plt.subplot(row, col + 1, i + 1)\n        plt.plot(df[f\"{column}\"], label=f\"{column}\")\n        plt.legend()\n        plt.xlabel(\"epochs\")\n        plt.ylabel(f\"{column}\")\n        plt.tight_layout(pad=2)  # padding\n\n    plt.subplot(row, col + 1, n + 1)\n    for column in df.columns:\n        plt.plot(df[f\"{column}\"], label=f\"{column}\")\n        plt.legend()\n    plt.xlabel(\"epochs\")","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:00:21.727375Z","iopub.execute_input":"2023-04-01T13:00:21.727816Z","iopub.status.idle":"2023-04-01T13:00:21.738010Z","shell.execute_reply.started":"2023-04-01T13:00:21.727760Z","shell.execute_reply":"2023-04-01T13:00:21.736673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_history(history)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:00:21.739951Z","iopub.execute_input":"2023-04-01T13:00:21.740447Z","iopub.status.idle":"2023-04-01T13:00:23.161904Z","shell.execute_reply.started":"2023-04-01T13:00:21.740383Z","shell.execute_reply":"2023-04-01T13:00:23.160786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference Model","metadata":{}},{"cell_type":"code","source":"class TFLiteModel(tf.keras.Model):\n    def __init__(self, model):\n        super().__init__()\n        self.prep_inputs = FeatureGen()\n        self.model = model\n        \n    @tf.function(input_signature=[tf.TensorSpec(shape=[None, 543, 3], dtype=tf.float32, name='inputs')])\n    def call(self, inputs):\n        x = self.prep_inputs(tf.cast(inputs, dtype=tf.float32))\n        x = tf.expand_dims(x, axis=0)\n        outputs = self.model(x)[0, :]\n        # Return a dictionary with the output tensor\n        return {'outputs': outputs}","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:32:01.596732Z","iopub.execute_input":"2023-04-01T13:32:01.597165Z","iopub.status.idle":"2023-04-01T13:32:01.607665Z","shell.execute_reply.started":"2023-04-01T13:32:01.597127Z","shell.execute_reply":"2023-04-01T13:32:01.606380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tflite_keras_model = TFLiteModel(model)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:32:02.722693Z","iopub.execute_input":"2023-04-01T13:32:02.723143Z","iopub.status.idle":"2023-04-01T13:32:02.736127Z","shell.execute_reply.started":"2023-04-01T13:32:02.723101Z","shell.execute_reply":"2023-04-01T13:32:02.734865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## TFlite model must perform inference with less than 100 milliseconds of latency per video on average","metadata":{}},{"cell_type":"code","source":"for i in range(2):\n    %timeit demo_output = tflite_keras_model(load_relevant_data_subset(path_train_df.path[i]))[\"outputs\"]","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:32:04.820774Z","iopub.execute_input":"2023-04-01T13:32:04.821161Z","iopub.status.idle":"2023-04-01T13:32:20.881204Z","shell.execute_reply.started":"2023-04-01T13:32:04.821126Z","shell.execute_reply":"2023-04-01T13:32:20.879716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tflite_keras_model.predict(load_relevant_data_subset(path_train_df.path[i]))[\"outputs\"].shape","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:32:20.883209Z","iopub.execute_input":"2023-04-01T13:32:20.883566Z","iopub.status.idle":"2023-04-01T13:32:21.148602Z","shell.execute_reply.started":"2023-04-01T13:32:20.883532Z","shell.execute_reply":"2023-04-01T13:32:21.147243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"converter = tf.lite.TFLiteConverter.from_keras_model(tflite_keras_model)\n\nconverter.target_spec.supported_ops = [\n    tf.lite.OpsSet.TFLITE_BUILTINS,\n    tf.lite.OpsSet.SELECT_TF_OPS,\n]\n# converter._experimental_lower_tensor_list_ops = False\n\ntflite_model = converter.convert()","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:32:26.011534Z","iopub.execute_input":"2023-04-01T13:32:26.011921Z","iopub.status.idle":"2023-04-01T13:32:37.842198Z","shell.execute_reply.started":"2023-04-01T13:32:26.011886Z","shell.execute_reply":"2023-04-01T13:32:37.840906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Save model\nmodel_path = \"model.tflite\"\nwith open(model_path, \"wb\") as f:\n    f.write(tflite_model)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:32:37.844463Z","iopub.execute_input":"2023-04-01T13:32:37.844911Z","iopub.status.idle":"2023-04-01T13:32:37.892237Z","shell.execute_reply.started":"2023-04-01T13:32:37.844871Z","shell.execute_reply":"2023-04-01T13:32:37.890794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model must use less than 40 MB of storage space","metadata":{}},{"cell_type":"code","source":"!ls -lh model.tflite","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:32:37.894252Z","iopub.execute_input":"2023-04-01T13:32:37.894752Z","iopub.status.idle":"2023-04-01T13:32:39.051095Z","shell.execute_reply.started":"2023-04-01T13:32:37.894698Z","shell.execute_reply":"2023-04-01T13:32:39.049836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission.zip $model_path","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:00:38.437418Z","iopub.execute_input":"2023-04-01T13:00:38.437772Z","iopub.status.idle":"2023-04-01T13:00:41.043028Z","shell.execute_reply.started":"2023-04-01T13:00:38.437737Z","shell.execute_reply":"2023-04-01T13:00:41.041682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%pip install tflite_runtime --quiet","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:00:41.045057Z","iopub.execute_input":"2023-04-01T13:00:41.045448Z","iopub.status.idle":"2023-04-01T13:00:53.446223Z","shell.execute_reply.started":"2023-04-01T13:00:41.045410Z","shell.execute_reply":"2023-04-01T13:00:53.444436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tflite_runtime.interpreter as tflite\n\n# load you tflite model\ninterpreter = tflite.Interpreter(\"model.tflite\")\nfound_signatures = list(interpreter.get_signature_list().keys())\n\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")\n\n# predict class of the relevant video\noutput = prediction_fn(inputs=load_relevant_data_subset(path_train_df.path[0]))\nsign = np.argmax(output[\"outputs\"])\n\nprint(\"PREDICTION : \", decoder(sign))\nprint(\"ACTUAL   : \", path_train_df.sign[0])","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:32:43.214882Z","iopub.execute_input":"2023-04-01T13:32:43.215282Z","iopub.status.idle":"2023-04-01T13:32:43.268358Z","shell.execute_reply.started":"2023-04-01T13:32:43.215247Z","shell.execute_reply":"2023-04-01T13:32:43.266997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nimport tensorflow as tf\nimport numpy as np\n\n# Load the TFLite model\ninterpreter = tf.lite.Interpreter(model_path='model.tflite')\ninterpreter.allocate_tensors()\n\n# Get input and output tensors information\ninput_details = interpreter.get_input_details()\noutput_details = interpreter.get_output_details()\n\n# Prepare your test input data (ensure it has the correct shape)\ninput_data = np.array(load_relevant_data_subset(path_train_df.path[0]), dtype=np.float32)\n\n# Set the input tensor with your test data\ninterpreter.set_tensor(input_details[0]['index'], input_data)\n\n# Run the model\ninterpreter.invoke()\n\n# Get the output tensor\noutput_data = interpreter.get_tensor(output_details[0]['index'])\nprint(output_data.shape)\nprint(output_data.argmax())\nprint(decoder(output_data.argmax()))\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2023-04-01T13:32:58.209050Z","iopub.execute_input":"2023-04-01T13:32:58.209469Z","iopub.status.idle":"2023-04-01T13:32:58.303491Z","shell.execute_reply.started":"2023-04-01T13:32:58.209431Z","shell.execute_reply":"2023-04-01T13:32:58.302033Z"}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}