{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Isolated Sign Language Recognition\n\n#### This notebook is based on the amazing [notebook](https://www.kaggle.com/code/lonnieqin/isolated-sign-language-recognition-with-dnn) by [Lonnie](https://www.kaggle.com/lonnieqin). I have used their dataloader pretty much as it is. Recommend to refer to their notebook for more details.\n\n## Transformers for the W\n\n⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜🟦🟦🟦🟦🟦🟦🟦🟦🟦🟦⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜\n⬛🟦🟦🟦🟦🟦🟦🟦🟦🟦⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜🟦🟦🟦🟦🟦🟦🟦🟦🟦⬛\n⬛🟦⬜⬜⬜⬜⬜⬜🟦⬜⬜🟦🟦🟦🟦🟦🟦🟦🟦⬜⬜🟦⬜⬜⬜⬜⬜⬜🟦⬛\n⬜⬛🟦🟦🟦🟦🟦🟦⬛⬜🟦🟦🟦🟦🟦🟦🟦🟦🟦🟦⬜⬛🟦🟦🟦🟦🟦🟦⬛⬜\n⬜⬛🟦🟦🟦🟦🟦🟦🟦⬛⬜⬜⬛⬛⬛⬛⬛⬛⬜⬜⬛🟦🟦🟦🟦🟦🟦🟦⬛⬜\n⬜⬛🟦🟦🟦🟦🟦⬛🏽🟦⬛⬜⬜⬛⬛⬛⬛⬜⬜⬛🟦🏽⬛🟦🟦🟦🟦🟦⬛⬜\n⬜⬜⬛🟦⬜⬜⬜⬜⬛🏽⬜⬛⬜⬜⬛⬛⬜⬜⬛⬜🏽⬛⬜⬜⬜⬜🟦⬛⬜⬜\n⬜⬜⬛🟦⬜⬜⬜⬛⬜⬛⬛⬜⬛⬜⬜⬜⬜⬛⬜⬛⬛⬜⬛⬜⬜⬜🟦⬛⬜⬜\n⬜⬜⬛🟦⬜⬜⬜⬜⬛🏽⬜⬜🟦⬛⬜⬜⬛🟦⬜⬜🏽⬛⬜⬜⬜⬜🟦⬛⬜⬜\n⬜⬜⬛🟦⬜⬜⬜⬜⬜⬛⬛⬜🟦⬛⬜⬜⬛🟦⬜⬛⬛⬜⬜⬜⬜⬜🟦⬛⬜⬜\n⬜⬜⬛⬛🟦🏽🏽🏽🏽🏽🏽🏽⬛⬛⬜⬜⬛⬛🏽🏽🏽🏽🏽🏽🏽🟦⬛⬛⬜⬜\n⬜⬜⬜⬛⬛🟦🟦🟦🟦🟦🟦🟦🟦⬛⬜⬜⬛🟦🟦🟦🟦🟦🟦🟦🟦⬛⬛⬜⬜⬜\n⬜⬜⬜⬛🟦⬛⬛⬛⬛⬛⬛⬛⬛⬛🟥🟥⬛⬛⬛⬛⬛⬛⬛⬛⬛🟦⬛⬜⬜⬜\n⬜⬜⬜⬛🟦🟦⬛⬛🟦⬛⬛⬛⬛⬛🟥🟥⬛⬛⬛⬛⬛🟦⬛⬛🟦🟦⬛⬜⬜⬜\n⬜⬜⬜⬛🟥🟦🟦⬛⬛⬛⬛⬛🟦⬛🟥🟥⬛🟦⬛⬛⬛⬛⬛🟦🟦🟥⬛⬜⬜⬜\n⬜⬜⬜⬜⬛🟥🟦🟦🟦⬛🟦🟦🟥⬛🟥🟥⬛🟥🟦🟦⬛🟦🟦🟦🟥⬛⬜⬜⬜⬜\n⬜⬜⬜⬜⬛🟥🟥🟦🟦⬛🟦🟦🟥⬛🟥🟥⬛🟥🟦🟦⬛🟦🟦🟥🟥⬛⬜⬜⬜⬜\n⬜⬜⬜⬜⬛🟥🟥🟥🟦⬛🟦🟦🟥⬛🟥🟥⬛🟥🟦🟦⬛🟦🟥🟥🟥⬛⬜⬜⬜⬜\n⬜⬜⬜⬜⬛🟥🟥🟥🟦⬛🟦🟦🟥⬛🟥🟥⬛🟥🟦🟦⬛🟦🟥🟥🟥⬛⬜⬜⬜⬜\n⬜⬜⬜⬜⬛🟥🟥🟥🟦⬛🟦🟦🟥⬛⬛⬛⬛🟥🟦🟦⬛🟦🟥🟥🟥⬛⬜⬜⬜⬜\n⬜⬜⬜⬜⬛⬜🟥🟥🟦⬛🟦🟦🟥🟥🟥🟥🟥🟥🟦🟦⬛🟦🟥🟥🟥⬛⬜⬜⬜⬜\n⬜⬜⬜⬜⬛🟦⬜🟥🟦⬛🟦🟦🟥⬜⬜⬜⬜🟥🟦🟦⬛🟦🟥⬜🟦⬛⬜⬜⬜⬜\n⬜⬜⬜⬜⬜⬛🟦⬜🟦⬛🟦⬜⬜⬛⬛⬛⬛⬜⬜🟦⬛🟦⬜🟦⬛⬜⬜⬜⬜⬜\n⬜⬜⬜⬜⬜⬜⬛🟦⬜⬛⬜⬜⬛⬜⬜⬜⬜⬛⬜⬜⬛⬜🟦⬛⬜⬜⬜⬜⬜⬜\n⬜⬜⬜⬜⬜⬜⬜⬛⬛⬛⬜⬜⬛⬜⬜⬜⬜⬛⬜⬜⬛⬛⬛⬜⬜⬜⬜⬜⬜⬜\n⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬛⬜⬜⬜⬜⬜⬜⬜⬜⬛⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜\n⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬛⬛⬜⬜⬜⬜⬛⬛⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜\n⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬛⬛⬛⬛⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜\n","metadata":{}},{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"!pip install wandb==0.13.10","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:15:42.678773Z","iopub.execute_input":"2023-02-27T10:15:42.679734Z","iopub.status.idle":"2023-02-27T10:15:57.484713Z","shell.execute_reply.started":"2023-02-27T10:15:42.679672Z","shell.execute_reply":"2023-02-27T10:15:57.482905Z"},"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    data_path = \"../input/asl-signs/\"\n    quick_experiment = False\n    is_training = True\n    use_aggregation_dataset = True\n    num_classes = 250\n    rows_per_frame = 543 ","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:15:57.489323Z","iopub.execute_input":"2023-02-27T10:15:57.490659Z","iopub.status.idle":"2023-02-27T10:15:57.501243Z","shell.execute_reply.started":"2023-02-27T10:15:57.490605Z","shell.execute_reply":"2023-02-27T10:15:57.499474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import Library","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow.keras as keras\nfrom tensorflow.keras import layers\nfrom tqdm import tqdm\nimport json\nimport os\nimport gc\nfrom sklearn.model_selection import train_test_split\nimport wandb\nfrom wandb.keras import WandbMetricsLogger, WandbModelCheckpoint, WandbCallback","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:15:57.504315Z","iopub.execute_input":"2023-02-27T10:15:57.505846Z","iopub.status.idle":"2023-02-27T10:15:57.514890Z","shell.execute_reply.started":"2023-02-27T10:15:57.505782Z","shell.execute_reply":"2023-02-27T10:15:57.512841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\n\nuser_secrets = UserSecretsClient()\nwandb_api = user_secrets.get_secret(\"wandb_api\") \n\nwandb.login(key=wandb_api)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:15:57.518489Z","iopub.execute_input":"2023-02-27T10:15:57.520401Z","iopub.status.idle":"2023-02-27T10:15:58.905023Z","shell.execute_reply.started":"2023-02-27T10:15:57.520315Z","shell.execute_reply":"2023-02-27T10:15:58.903440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utilities","metadata":{}},{"cell_type":"code","source":"\nSEED = 42\nos.environ[\"TF_CUDNN_DETERMINISTIC\"] = \"1\"\n# keras.utils.set_random_seed(SEED)\n\nBATCH_SIZE = 32\nAUTO = tf.data.AUTOTUNE\nINPUT_SHAPE = (543,3)\nNUM_CLASSES = 250\n\n# OPTIMIZER\nLEARNING_RATE = 1e-3\nWEIGHT_DECAY = 1e-5\n\n# TRAINING\nEPOCHS = 100\n\n# TUBELET EMBEDDING\nPATCH_SIZE = 3\nNUM_PATCHES = (INPUT_SHAPE[0] // PATCH_SIZE) ** 2\nNATURE = \"NOVEL\"\n\n# ViViT ARCHITECTURE\nLAYER_NORM_EPS = 1e-6\nPROJECTION_DIM = 64\nNUM_HEADS = 4\nNUM_LAYERS = 1\n\nconfig = {\n    \"batch_size\": BATCH_SIZE,\n    \"learning_rate\": LEARNING_RATE,\n    \"weight_decay\": WEIGHT_DECAY,\n    \"epochs\": EPOCHS,\n    \"patch_size\": PATCH_SIZE,\n    \"num_patches\": NUM_PATCHES,\n    \"projection_dim\": PROJECTION_DIM,\n    \"num_heads\": NUM_HEADS,\n    \"num_layers\": NUM_LAYERS,\n    \"nature\": NATURE,\n    \"seed\": SEED,\n}\n","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:15:58.909872Z","iopub.execute_input":"2023-02-27T10:15:58.911913Z","iopub.status.idle":"2023-02-27T10:15:58.924141Z","shell.execute_reply.started":"2023-02-27T10:15:58.911843Z","shell.execute_reply":"2023-02-27T10:15:58.922356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wandb.init(project=\"asl\", entity=\"Prajwal31\", config = config)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:15:58.927442Z","iopub.execute_input":"2023-02-27T10:15:58.928068Z","iopub.status.idle":"2023-02-27T10:16:31.419120Z","shell.execute_reply.started":"2023-02-27T10:15:58.928010Z","shell.execute_reply":"2023-02-27T10:16:31.417549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_relevant_data_subset_with_imputation(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    data.replace(np.nan, 0, inplace=True)\n    n_frames = int(len(data) / CFG.rows_per_frame)\n    data = data.values.reshape(n_frames, CFG.rows_per_frame, len(data_columns))\n    return data.astype(np.float32)\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / CFG.rows_per_frame)\n    data = data.values.reshape(n_frames, CFG.rows_per_frame, len(data_columns))\n    return data.astype(np.float32)\n\ndef read_dict(file_path):\n    path = os.path.expanduser(file_path)\n    with open(path, \"r\") as f:\n        dic = json.load(f)\n    return dic","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:16:31.426706Z","iopub.execute_input":"2023-02-27T10:16:31.430827Z","iopub.status.idle":"2023-02-27T10:16:31.452573Z","shell.execute_reply.started":"2023-02-27T10:16:31.430747Z","shell.execute_reply":"2023-02-27T10:16:31.450695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load data","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(f\"{CFG.data_path}train.csv\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:16:31.455541Z","iopub.execute_input":"2023-02-27T10:16:31.456698Z","iopub.status.idle":"2023-02-27T10:16:31.715524Z","shell.execute_reply.started":"2023-02-27T10:16:31.456631Z","shell.execute_reply":"2023-02-27T10:16:31.713908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 21 participants. Each of them create about 3000 to 5000 training records.","metadata":{}},{"cell_type":"code","source":"train.participant_id.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:16:31.723049Z","iopub.execute_input":"2023-02-27T10:16:31.727249Z","iopub.status.idle":"2023-02-27T10:16:31.747325Z","shell.execute_reply.started":"2023-02-27T10:16:31.727154Z","shell.execute_reply":"2023-02-27T10:16:31.745020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.participant_id.value_counts().plot(kind=\"bar\")","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:16:31.749340Z","iopub.execute_input":"2023-02-27T10:16:31.750313Z","iopub.status.idle":"2023-02-27T10:16:32.382169Z","shell.execute_reply.started":"2023-02-27T10:16:31.750254Z","shell.execute_reply":"2023-02-27T10:16:32.380502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 94477 training samples in total.","metadata":{}},{"cell_type":"code","source":"len(train)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:16:32.384755Z","iopub.execute_input":"2023-02-27T10:16:32.386175Z","iopub.status.idle":"2023-02-27T10:16:32.406716Z","shell.execute_reply.started":"2023-02-27T10:16:32.386092Z","shell.execute_reply":"2023-02-27T10:16:32.404580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are 250 kinds of sign languages that we need to make prediction on. Each kind of sign languages contains about 300 to 400 samples.","metadata":{}},{"cell_type":"code","source":"label_index = read_dict(f\"{CFG.data_path}sign_to_prediction_index_map.json\")\nindex_label = dict([(label_index[key], key) for key in label_index])\nprint(label_index)\ntrain[\"label\"] = train[\"sign\"].map(lambda sign: label_index[sign])\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:16:32.409889Z","iopub.execute_input":"2023-02-27T10:16:32.417047Z","iopub.status.idle":"2023-02-27T10:16:32.532974Z","shell.execute_reply.started":"2023-02-27T10:16:32.416964Z","shell.execute_reply":"2023-02-27T10:16:32.531126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"sign\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:16:32.535428Z","iopub.execute_input":"2023-02-27T10:16:32.537098Z","iopub.status.idle":"2023-02-27T10:16:32.572068Z","shell.execute_reply.started":"2023-02-27T10:16:32.537029Z","shell.execute_reply":"2023-02-27T10:16:32.570228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modeling\nI am still exploring how to handle this dataset. In order to make it easy to start with and train faster, I use mean frame as training input data.","metadata":{}},{"cell_type":"code","source":"if CFG.is_training:\n    if CFG.use_aggregation_dataset == False:\n        xs = []\n        ys = []\n        num_frames = np.zeros(len(train))\n        for i in tqdm(range(len(train))):\n            path = f\"{CFG.data_path}{train.iloc[i].path}\"\n            data = load_relevant_data_subset_with_imputation(path)\n            ## Mean Aggregation\n            xs.append(np.mean(data, axis=0))\n            ys.append(train.iloc[i].label)\n            num_frames[i] = data.shape[0]\n            if CFG.quick_experiment and i == 4999:\n                break\n        ## Save number of frames of each training sample for data analysis\n        train[\"num_frames\"] = num_frames\n        X = np.array(xs)\n        y = np.array(ys)\n        print(train[\"num_frames\"].describe())\n        train.to_csv(\"train.csv\", index=False)\n    else:\n        X = np.load(\"/kaggle/input/isolated-sign-language-aggregation-dataset/X.npy\")\n        y = np.load(\"/kaggle/input/isolated-sign-language-aggregation-dataset/y.npy\")\n    print(X.shape, y.shape)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:16:32.579880Z","iopub.execute_input":"2023-02-27T10:16:32.581676Z","iopub.status.idle":"2023-02-27T10:16:32.772858Z","shell.execute_reply.started":"2023-02-27T10:16:32.581608Z","shell.execute_reply":"2023-02-27T10:16:32.771152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_transformer(\n    input_shape,\n    num_classes,\n    embed_dim,\n    num_heads,\n    layer_norm_eps,\n    transformer_layers,\n    tubelet_embedder,\n    positional_encoder,\n):\n\n    inputs = keras.Input(shape=input_shape)\n    encoded_patches = tubelet_embedder(inputs)\n    encoded_patches  = positional_encoder(encoded_patches)\n\n    # Create multiple layers of the Transformer block.\n    for _ in range(transformer_layers):\n    # Layer normalization and MHSA\n        x1 = layers.LayerNormalization(epsilon=1e-6)(encoded_patches)\n        attention_output = layers.MultiHeadAttention(\n          num_heads=num_heads, key_dim=embed_dim // num_heads, dropout=0.1\n        )(x1, x1)\n\n        # Skip connection\n        x2 = layers.Add()([attention_output, encoded_patches])\n\n        # Layer Normalization and MLP\n        x3 = layers.LayerNormalization(epsilon=1e-6)(x2)\n        x3 = keras.Sequential(\n          [\n              layers.Dense(units=embed_dim * 4, activation=tf.nn.gelu),\n              layers.Dense(units=embed_dim, activation=tf.nn.gelu),\n          ]\n        )(x3)\n\n        # Skip connection\n        encoded_patches = layers.Add()([x3, x2])\n\n    # Layer normalization and Global average pooling.\n    representation = layers.LayerNormalization(epsilon=layer_norm_eps)(encoded_patches)\n    representation = layers.GlobalAvgPool1D()(representation)\n    representation = layers.Dense(512, activation = tf.nn.gelu)(representation)\n    representation = layers.Dense(256, activation = tf.nn.gelu)(representation)\n\n    # Classify outputs.\n    outputs = layers.Dense(units=num_classes, activation=\"softmax\")(representation)\n\n    # Create the Keras model.\n    model = keras.Model(inputs=inputs, outputs=outputs)\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:16:32.779811Z","iopub.execute_input":"2023-02-27T10:16:32.784484Z","iopub.status.idle":"2023-02-27T10:16:32.808676Z","shell.execute_reply.started":"2023-02-27T10:16:32.784405Z","shell.execute_reply":"2023-02-27T10:16:32.806122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TubletEmbedding(layers.Layer):\n    '''\n    Embedd a video into a sequence of patches using short and long term convolutions\n    like inception stlye convolutions with multiple kernel sizes\n    '''\n    def __init__(self, embed_dim, patch_size, **kwargs):\n        super().__init__(**kwargs)\n        self.filters = []\n        for i in range(1, 4):\n            proj = layers.Conv1D(\n                filters=embed_dim,\n                kernel_size=(patch_size * i),\n                strides=(patch_size),\n                padding=\"SAME\",\n            )\n            self.filters.append(proj)\n        self.flatten = layers.Reshape(target_shape=(-1, embed_dim))\n    \n    def call(self, videos):\n        projected_patches = []\n        for proj in self.filters:\n            projected_patches.append(proj(videos))\n        projected_patches = tf.concat(projected_patches, axis=-1)\n        flattened_patches = self.flatten(projected_patches)\n        return flattened_patches\n    \nclass PositionalEncoder(layers.Layer):\n    def __init__(self, embed_dim, **kwargs):\n        super().__init__(**kwargs)\n        self.embed_dim = embed_dim\n\n    def build(self, input_shape):\n        _, num_tokens, _ = input_shape\n        self.position_embedding = layers.Embedding(\n            input_dim=num_tokens, output_dim=self.embed_dim\n        )\n        self.positions = tf.range(start=0, limit=num_tokens, delta=1)\n\n    def call(self, encoded_tokens):\n        # Encode the positions and add it to the encoded tokens\n        encoded_positions = self.position_embedding(self.positions)\n        encoded_tokens = encoded_tokens + encoded_positions\n        return encoded_tokens","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:16:32.819316Z","iopub.execute_input":"2023-02-27T10:16:32.820659Z","iopub.status.idle":"2023-02-27T10:16:32.842274Z","shell.execute_reply.started":"2023-02-27T10:16:32.820592Z","shell.execute_reply":"2023-02-27T10:16:32.840354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    \n    tubelet_embedder = TubletEmbedding(PROJECTION_DIM, PATCH_SIZE)\n    positional_encoder = PositionalEncoder(PROJECTION_DIM)\n    model = get_transformer(\n        input_shape=INPUT_SHAPE,\n        num_classes=NUM_CLASSES,\n        embed_dim=PROJECTION_DIM,\n        num_heads=NUM_HEADS,\n        layer_norm_eps=LAYER_NORM_EPS,\n        transformer_layers=NUM_LAYERS,\n        tubelet_embedder=tubelet_embedder,\n        positional_encoder=positional_encoder,\n    )\n    optimizer = keras.optimizers.Adam(learning_rate=LEARNING_RATE)\n    model.compile(\n        optimizer=optimizer,\n        loss=tf.keras.losses.SparseCategoricalCrossentropy(), \n        metrics=[\n            \"accuracy\", \n            tf.keras.metrics.SparseTopKCategoricalAccuracy(k=10)\n        ],\n        steps_per_execution=32\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:16:32.845285Z","iopub.execute_input":"2023-02-27T10:16:32.847040Z","iopub.status.idle":"2023-02-27T10:16:32.864481Z","shell.execute_reply.started":"2023-02-27T10:16:32.846972Z","shell.execute_reply":"2023-02-27T10:16:32.860941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if CFG.is_training:\n    X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42, stratify=y)\n    print(X_train.shape, y_train.shape, X_val.shape, y_val.shape)\n    del X, y\n    gc.collect()\n    model = get_model()\n#     model.load_weights(\"/kaggle/input/isoasl/model.h5\")\n    \n#     save_locally = tf.saved_model.SaveOptions(experimental_io_device='/job:localhost')\n    callbacks = [tf.keras.callbacks.ModelCheckpoint(\"model.h5\"), WandbMetricsLogger(log_freq=5), \n                 WandbCallback()]\n#     callbacks = [tf.keras.callbacks.ModelCheckpoint(\"model.h5\", options=save_locally)]\n#     callbacks = [WandbCallback()]\n    model.fit(\n        X_train, \n        y_train, \n        epochs=50, \n        validation_data=(X_val, y_val), \n        batch_size=BATCH_SIZE, \n        callbacks=callbacks\n    )\nelse:\n    model = get_model()\n    model.load_weights(\"/kaggle/input/isoasl/model.h5\")\n#     model = tf.keras.models.load_model(\"/kaggle/input/sign-language-prediction-model/model.h5\")\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:16:32.867791Z","iopub.execute_input":"2023-02-27T10:16:32.869190Z","iopub.status.idle":"2023-02-27T10:19:19.401505Z","shell.execute_reply.started":"2023-02-27T10:16:32.869127Z","shell.execute_reply":"2023-02-27T10:19:19.400184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wandb.finish()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:19:19.403136Z","iopub.execute_input":"2023-02-27T10:19:19.404038Z","iopub.status.idle":"2023-02-27T10:19:25.940008Z","shell.execute_reply.started":"2023-02-27T10:19:19.403949Z","shell.execute_reply":"2023-02-27T10:19:25.938257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.save(\"model.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:19:25.942460Z","iopub.execute_input":"2023-02-27T10:19:25.943109Z","iopub.status.idle":"2023-02-27T10:19:25.950148Z","shell.execute_reply.started":"2023-02-27T10:19:25.943048Z","shell.execute_reply":"2023-02-27T10:19:25.948125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Model for inference","metadata":{}},{"cell_type":"code","source":"def get_inference_model(model):\n    inputs = tf.keras.Input((543, 3), dtype=tf.float32, name=\"inputs\")\n    x = tf.where(tf.math.is_nan(inputs), tf.zeros_like(inputs), inputs)\n    x = tf.reduce_mean(x, axis=0, keepdims=True)\n    model.layers.pop(0)\n    x = model(x)\n    output = tf.keras.layers.Activation(activation=\"linear\", name=\"outputs\")(x)\n    inference_model = tf.keras.Model(inputs=inputs, outputs=output) \n    inference_model.compile(loss=tf.keras.losses.SparseCategoricalCrossentropy(), metrics=[\"accuracy\"])\n    return inference_model","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:19:25.953335Z","iopub.execute_input":"2023-02-27T10:19:25.954362Z","iopub.status.idle":"2023-02-27T10:19:25.969829Z","shell.execute_reply.started":"2023-02-27T10:19:25.954063Z","shell.execute_reply":"2023-02-27T10:19:25.967769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inference_model = get_inference_model(model)\ninference_model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:19:25.973056Z","iopub.execute_input":"2023-02-27T10:19:25.973775Z","iopub.status.idle":"2023-02-27T10:19:26.247646Z","shell.execute_reply.started":"2023-02-27T10:19:25.973692Z","shell.execute_reply":"2023-02-27T10:19:26.246467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create submission file","metadata":{}},{"cell_type":"code","source":"converter = tf.lite.TFLiteConverter.from_keras_model(inference_model)\ntflite_model = converter.convert()\nmodel_path = \"model.tflite\"\n# Save the model.\nwith open(model_path, 'wb') as f:\n    f.write(tflite_model)","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:19:26.249143Z","iopub.execute_input":"2023-02-27T10:19:26.249929Z","iopub.status.idle":"2023-02-27T10:19:36.957130Z","shell.execute_reply.started":"2023-02-27T10:19:26.249885Z","shell.execute_reply":"2023-02-27T10:19:36.955319Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission.zip $model_path","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:19:36.959236Z","iopub.execute_input":"2023-02-27T10:19:36.959792Z","iopub.status.idle":"2023-02-27T10:19:38.240937Z","shell.execute_reply.started":"2023-02-27T10:19:36.959736Z","shell.execute_reply":"2023-02-27T10:19:38.239138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Making Prediction","metadata":{}},{"cell_type":"code","source":"!pip install tflite-runtime","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:19:38.243753Z","iopub.execute_input":"2023-02-27T10:19:38.244326Z","iopub.status.idle":"2023-02-27T10:19:53.446527Z","shell.execute_reply.started":"2023-02-27T10:19:38.244269Z","shell.execute_reply":"2023-02-27T10:19:53.444809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The performance is not optimal so far. However it can make correct prediction sometimes.","metadata":{}},{"cell_type":"code","source":"import tflite_runtime.interpreter as tflite\ninterpreter = tflite.Interpreter(model_path)\nfound_signatures = list(interpreter.get_signature_list().keys())\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")\nfor i in range(100):\n    frames = load_relevant_data_subset(f'/kaggle/input/asl-signs/{train.iloc[i].path}')\n    output = prediction_fn(inputs=frames)\n    sign = np.argmax(output[\"outputs\"])\n    print(f\"Predicted label: {index_label[sign]}, Actual Label: {train.iloc[i].sign}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-27T10:19:53.450607Z","iopub.execute_input":"2023-02-27T10:19:53.451230Z","iopub.status.idle":"2023-02-27T10:19:57.756306Z","shell.execute_reply.started":"2023-02-27T10:19:53.451167Z","shell.execute_reply":"2023-02-27T10:19:57.753368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}