{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"CV## Configuration","metadata":{}},{"cell_type":"markdown","source":"## Import Packages","metadata":{}},{"cell_type":"code","source":"from tensorflow.python.lib.io import file_io\nfrom io import StringIO\nfrom io import BytesIO\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nimport os\nfrom sklearn.utils import shuffle\nimport gc\nfrom tensorflow.keras import regularizers\nfrom tensorflow.keras.utils import Sequence\nimport math\nfrom tensorflow.keras.callbacks import ModelCheckpoint\n","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:11:20.191401Z","iopub.execute_input":"2023-09-14T16:11:20.192213Z","iopub.status.idle":"2023-09-14T16:11:29.597436Z","shell.execute_reply.started":"2023-09-14T16:11:20.192168Z","shell.execute_reply":"2023-09-14T16:11:29.596266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    aggregation_data_path = \"../input/isolated-sign-language-aggregation-dataset/\"\n    data_path = \"../input/asl-signs/\"\n    use_aggregation_dataset = True\n    num_classes = 250\n    rows_per_frame = 543 ","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:11:29.599908Z","iopub.execute_input":"2023-09-14T16:11:29.601114Z","iopub.status.idle":"2023-09-14T16:11:29.607815Z","shell.execute_reply.started":"2023-09-14T16:11:29.601070Z","shell.execute_reply":"2023-09-14T16:11:29.606739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utilities","metadata":{}},{"cell_type":"code","source":"def load_relevant_data_subset_with_imputation(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    data_np = data.values.astype(np.float32)\n    data_np[np.isnan(data_np)] = 0  # Replace NaN values with 0\n    n_frames = int(len(data) / CFG.rows_per_frame)\n    data = data_np.reshape(n_frames, CFG.rows_per_frame, len(data_columns))\n    return data\n\ndef read_dict(file_path):\n    path = os.path.expanduser(file_path)\n    with open(path, \"r\") as f:\n        dic = json.load(f)\n    return dic\n# Helpers\n\nROWS_PER_FRAME = 543\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:11:29.610364Z","iopub.execute_input":"2023-09-14T16:11:29.610763Z","iopub.status.idle":"2023-09-14T16:11:29.626363Z","shell.execute_reply.started":"2023-09-14T16:11:29.610724Z","shell.execute_reply":"2023-09-14T16:11:29.625125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(f\"{CFG.aggregation_data_path}train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:11:29.630953Z","iopub.execute_input":"2023-09-14T16:11:29.631243Z","iopub.status.idle":"2023-09-14T16:11:29.857399Z","shell.execute_reply.started":"2023-09-14T16:11:29.631215Z","shell.execute_reply":"2023-09-14T16:11:29.856339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:11:29.859093Z","iopub.execute_input":"2023-09-14T16:11:29.859453Z","iopub.status.idle":"2023-09-14T16:11:29.866383Z","shell.execute_reply.started":"2023-09-14T16:11:29.859420Z","shell.execute_reply":"2023-09-14T16:11:29.865303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"LIPS_IDXS0 = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\nLEFT_HAND_IDXS0  = np.arange(468,489)\nRIGHT_HAND_IDXS0 = np.arange(522,543)\nPOSE_IDXS0       = np.arange(502, 512)\nLANDMARK_IDXS0   = np.concatenate((LIPS_IDXS0, LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0, POSE_IDXS0))\n\n\ndef center_x_coordinate(video):\n    video[:, :, 0] += -0.5\n    return video\n    \ndef discard_z_column(video):\n    return video[:, :, (0, 1)]\n\ndef filter_landmarks(video):\n    return video[:, LANDMARK_IDXS0, :]\n\ndef create_missing_coordinate_columns(video):\n    # creates extra binary columns carrying the information of where the landmark missing:\n    # [x, y, z, landmark_exists]\n    nans = np.isnan(video[:, :, 0])\n    \n    extra_column_array = np.empty((video.shape[0], video.shape[1], video.shape[2] + 1), dtype=video.dtype)\n    extra_column_array[:, :, :video.shape[2]] = video\n    extra_column_array[:, :, video.shape[2]] = np.logical_not(nans)\n    return extra_column_array\n\ndef replace_nans_with_value(video, replace_value):\n    return np.nan_to_num(video, nan=replace_value)\n\ndef pad_frames(video, frame_size, pad_value):\n    \n    padding_config = ((0, frame_size - video.shape[0]), (0, 0), (0, 0))\n    \n    padded_data = np.pad(video, padding_config, mode='constant', constant_values=pad_value)\n    return padded_data\n\ndef limit_video_to_max_x_frames(video: np.ndarray, max_frames: int):\n    no_frames = video.shape[0]\n    if no_frames <= max_frames:\n        return video\n    \n    new_frames = np.linspace(0, no_frames - 1, max_frames, dtype=np.int32)\n    return video[new_frames]","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:11:29.868125Z","iopub.execute_input":"2023-09-14T16:11:29.868859Z","iopub.status.idle":"2023-09-14T16:11:29.887310Z","shell.execute_reply.started":"2023-09-14T16:11:29.868813Z","shell.execute_reply":"2023-09-14T16:11:29.886243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_video(video):\n    video = limit_video_to_max_x_frames(video, 30)\n    video = center_x_coordinate(video)\n    video = discard_z_column(video)\n    video = filter_landmarks(video)\n    #video = create_missing_coordinate_columns(video)\n    #video = replace_nans_with_value(video, -2)\n    video = pad_frames(video, 30, -42)\n\n    return video","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:11:29.889105Z","iopub.execute_input":"2023-09-14T16:11:29.889480Z","iopub.status.idle":"2023-09-14T16:11:29.902985Z","shell.execute_reply.started":"2023-09-14T16:11:29.889440Z","shell.execute_reply":"2023-09-14T16:11:29.901852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.use_aggregation_dataset == True:\n    xs = []\n    ys = []\n    num_frames = np.zeros(len(train))\n    for i in tqdm(range(len(train))):\n        #if train.iloc[i].num_frames > 40: continue\n        path = f\"{CFG.data_path}{train.iloc[i].path}\"\n        data = load_relevant_data_subset_with_imputation(path)\n        data = preprocess_video(data)\n        xs.append(data)\n        ys.append(train.iloc[i].label)\n        num_frames[i] = data.shape[0]\n        if i >= 30000:\n            break\n    ## Save number of frames of each training sample for data analysis\n    train[\"num_frames\"] = num_frames\n    X = np.array(xs)\n    y = np.array(ys)\n    print(train[\"num_frames\"].describe())\n    train.to_csv(\"train.csv\", index=False)\n    \n    #os.mkdir(CFG.agg_data)\n    #np.save(f\"{CFG.agg_data}X\", X)\n    #np.save(f\"{CFG.agg_data}Y\", y)\n    print(X.shape, y.shape)\n    del xs\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:11:29.904700Z","iopub.execute_input":"2023-09-14T16:11:29.905092Z","iopub.status.idle":"2023-09-14T16:11:56.360228Z","shell.execute_reply.started":"2023-09-14T16:11:29.905052Z","shell.execute_reply":"2023-09-14T16:11:56.359065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# MODEL","metadata":{}},{"cell_type":"code","source":"def create_convlstm1d_model(input_shape):\n    input_layer = layers.Input(shape=input_shape)\n\n    x = layers.Masking(mask_value=-42.)(input_layer)\n    x = layers.TimeDistributed(layers.Dense(12, activation='relu'))(x)\n    x = layers.TimeDistributed(layers.Dense(24, activation='relu'))(x)\n    x = layers.TimeDistributed(layers.Dense(48, activation='relu'))(x)\n    x = layers.TimeDistributed(layers.Dense(96, activation='relu'))(x)\n    x = layers.Conv1D(128, kernel_size=3, activation='relu', padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.TimeDistributed(layers.Dropout(0.2))(x)\n    x = layers.TimeDistributed(layers.Flatten())(x)\n    x = layers.LSTM(32, return_sequences=True)(x)  # New LSTM layer added\n    x = layers.Bidirectional(layers.LSTM(32, return_sequences=True))(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.Dropout(0.2)(x)\n    x = layers.Conv1D(64, kernel_size=3, activation='relu', padding='same')(x)\n    x = layers.BatchNormalization()(x)\n    x = layers.GlobalMaxPooling1D()(x)\n    x = layers.Dropout(0.2)(x)\n\n    model = models.Model(inputs=input_layer, outputs=x)\n    return model\n\n\ndef create_combined_convlstm1d_model(lips_model, left_hand_model, right_hand_model, pose_model):\n    sequence_input = layers.Input(shape=(None, 92, 2))\n    lips_input = tf.gather(sequence_input, LIPS, axis=2)\n    left_hand_input = tf.gather(sequence_input, LEFT_HAND, axis=2)\n    right_hand_input = tf.gather(sequence_input, RIGHT_HAND, axis=2)\n    pose_input = tf.gather(sequence_input, POSE, axis=2)\n\n    lips_features = lips_model(lips_input)\n    left_hand_features = left_hand_model(left_hand_input)\n    right_hand_features = right_hand_model(right_hand_input)\n    pose_features = pose_model(pose_input)\n    \n    hand_features = layers.Average()([left_hand_features, right_hand_features])\n    \n    lips_weight = tf.keras.backend.variable(0.2, name='lips_weight')\n    hand_weight = tf.keras.backend.variable(0.6, name='hand_weight')\n    pose_weight = tf.keras.backend.variable(0.2, name='pose_weight')\n    \n    weighted_lips_features = layers.Multiply()([lips_features, tf.fill(tf.shape(lips_features), lips_weight)])\n    weighted_hand_features = layers.Multiply()([hand_features, tf.fill(tf.shape(hand_features), hand_weight)])\n    weighted_pose_features = layers.Multiply()([pose_features, tf.fill(tf.shape(pose_features), pose_weight)])\n    \n\n    combined_features = layers.Add()([weighted_lips_features, weighted_hand_features, weighted_pose_features])\n\n    #combined_features = layers.Concatenate(axis=-1)([lips_features, hand_features, pose_features])\n    \n    #combined_features = layers.Dense(128, activation='relu')(combined_features)\n    #combined_features = layers.Dropout(0.3)(combined_features)\n    \n    \n    output = layers.Dense(250, activation='softmax')(combined_features)\n    \n    model = models.Model(inputs=sequence_input, outputs=output)\n    model.compile(optimizer='adam', loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n    return model\n\nLIPS       = np.arange(0,40)\nLEFT_HAND  = np.arange(40,61)\nRIGHT_HAND = np.arange(61,82)\nPOSE       = np.arange(82, 92)\nLANDMARK   = np.concatenate((LIPS, LEFT_HAND, RIGHT_HAND, POSE))\n\n\n#l2_lambda = 0.0004  # Choose an appropriate value for L2 regularization\n\nlips_convlstm1d_model = create_convlstm1d_model((None, len(LIPS), 2))\nleft_hand_convlstm1d_model = create_convlstm1d_model((None, len(LEFT_HAND), 2))\nright_hand_convlstm1d_model = create_convlstm1d_model((None, len(RIGHT_HAND), 2))\npose_convlstm1d_model = create_convlstm1d_model((None, len(POSE), 2))\n\n\ncombined_convlstm1d_model = create_combined_convlstm1d_model(lips_convlstm1d_model, left_hand_convlstm1d_model, right_hand_convlstm1d_model, pose_convlstm1d_model)\ncombined_convlstm1d_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:11:56.362071Z","iopub.execute_input":"2023-09-14T16:11:56.362948Z","iopub.status.idle":"2023-09-14T16:12:07.356322Z","shell.execute_reply.started":"2023-09-14T16:11:56.362904Z","shell.execute_reply":"2023-09-14T16:12:07.355472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\nX_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:12:07.360774Z","iopub.execute_input":"2023-09-14T16:12:07.361148Z","iopub.status.idle":"2023-09-14T16:12:07.392274Z","shell.execute_reply.started":"2023-09-14T16:12:07.361106Z","shell.execute_reply":"2023-09-14T16:12:07.391185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del X\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:12:07.394214Z","iopub.execute_input":"2023-09-14T16:12:07.394982Z","iopub.status.idle":"2023-09-14T16:12:07.626694Z","shell.execute_reply.started":"2023-09-14T16:12:07.394941Z","shell.execute_reply":"2023-09-14T16:12:07.625664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.callbacks import ModelCheckpoint\n\n\n# Define the model checkpoint\ncheckpoint_filepath = 'best_model.h5'\nmodel_checkpoint = ModelCheckpoint(\n    filepath=checkpoint_filepath,\n    monitor='val_loss',\n    save_best_only=True,\n    mode='min',\n    verbose=1\n)\n\n# Add the checkpoint callback to the list of callbacks\ncallbacks = [model_checkpoint]\n","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:12:07.628395Z","iopub.execute_input":"2023-09-14T16:12:07.628724Z","iopub.status.idle":"2023-09-14T16:12:07.634762Z","shell.execute_reply.started":"2023-09-14T16:12:07.628694Z","shell.execute_reply":"2023-09-14T16:12:07.633471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_epochs = 10\nbatch_size = 512\nhistory = combined_convlstm1d_model.fit(\n    X_train, y_train,\n    epochs=num_epochs,\n    batch_size=batch_size,\n    validation_data=(X_test, y_test)\n)","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:12:07.636607Z","iopub.execute_input":"2023-09-14T16:12:07.637445Z","iopub.status.idle":"2023-09-14T16:13:15.475877Z","shell.execute_reply.started":"2023-09-14T16:12:07.637404Z","shell.execute_reply":"2023-09-14T16:13:15.474376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_inference_model(model):\n    def preprocessing_function(video):\n        frames = 30\n        video = tf_replace_nans_with_zero(video)\n        video = tf_limit_video_to_max_x_frames(video, frames)\n        video = tf_center_x_coordinate(video)\n        video = tf_discard_z_column(video)\n        video = tf_filter_landmarks(video)\n        video = tf_pad_frames(video, frames, -42)\n        return video\n    \n    inputs = tf.keras.Input((543, 3), dtype=tf.float32, name=\"inputs\")\n    x = tf.keras.layers.Lambda(preprocessing_function)(inputs)\n    x = tf.expand_dims(x, axis=0, name=\"FakeBatchDim\")\n    x = model(x)\n    output = tf.keras.layers.Activation(activation=\"linear\", name=\"outputs\")(x)\n    inference_model = tf.keras.Model(inputs=inputs, outputs=output) \n    inference_model.compile(loss=tf.keras.losses.SparseCategoricalCrossentropy(), metrics=[\"accuracy\"])\n    return inference_model\n\n# ---\n\ndef tf_replace_nans_with_zero(video: tf.Tensor) -> tf.Tensor:\n    return tf.where(tf.math.is_nan(video), tf.zeros_like(video), video)\n\ndef tf_limit_video_to_max_x_frames(video: tf.Tensor, max_frames: int) -> tf.Tensor:\n    no_frames = tf.shape(video)[0]\n    return tf.cond(\n        tf.less_equal(no_frames, max_frames),\n        lambda: video,\n        lambda: tf.gather(video, tf.cast(tf.linspace(0.0, tf.cast(no_frames - 1, tf.float32), max_frames), dtype=tf.int32), axis=0)\n    )\n\ndef tf_center_x_coordinate(video: tf.Tensor) -> tf.Tensor:\n    input_shape = tf.shape(video)\n    \n    slice_index = [slice(None), slice(None), 0]\n    last_values = video[..., 0]\n    new_last_values = last_values - 0.5\n    new_video = tf.keras.backend.concatenate([tf.expand_dims(new_last_values, -1), video[..., 1:]], axis=-1)\n\n    return new_video\n\ndef tf_discard_z_column(video: tf.Tensor) -> tf.Tensor:\n    return video[:, :, slice(0, 2)]\n\ndef tf_filter_landmarks(video: tf.Tensor) -> tf.Tensor:\n    return tf.gather(video, LANDMARK_IDXS0, axis=1)\n\ndef tf_pad_frames(video: tf.Tensor, frame_size: int, pad_value: int) -> tf.Tensor:\n    padding_config = ((0, frame_size - tf.shape(video)[0]), (0, 0), (0, 0))\n    padded_data = tf.pad(video, padding_config, mode='CONSTANT', constant_values=pad_value)\n    return padded_data","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:13:15.477826Z","iopub.execute_input":"2023-09-14T16:13:15.478475Z","iopub.status.idle":"2023-09-14T16:13:15.498518Z","shell.execute_reply.started":"2023-09-14T16:13:15.478431Z","shell.execute_reply":"2023-09-14T16:13:15.497323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inf_model = get_inference_model(combined_convlstm1d_model)\ninf_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:13:15.500329Z","iopub.execute_input":"2023-09-14T16:13:15.500727Z","iopub.status.idle":"2023-09-14T16:13:19.675442Z","shell.execute_reply.started":"2023-09-14T16:13:15.500688Z","shell.execute_reply":"2023-09-14T16:13:19.674422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"converter = tf.lite.TFLiteConverter.from_keras_model(inf_model)\n\n\n# converter._experimental_lower_tensor_list_ops = False\n\ntflite_model = converter.convert()\n\nmodel_path = \"model.tflite\"\n\n# Save the model.\nwith open(model_path, 'wb') as f:\n    f.write(tflite_model)\n    \n!zip submission.zip $model_path","metadata":{"execution":{"iopub.status.busy":"2023-09-14T16:13:19.676572Z","iopub.execute_input":"2023-09-14T16:13:19.677738Z","iopub.status.idle":"2023-09-14T16:14:52.768419Z","shell.execute_reply.started":"2023-09-14T16:13:19.677696Z","shell.execute_reply":"2023-09-14T16:14:52.767028Z"},"trusted":true},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}