{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"##### DNN for Hand Recognition\n\nTODO: \n1. Normalize the data over the feature columns \n\n```\ndef normalizeData(arr):\n  stdArr = np.std(arr)\n  meanArr = np.mean(arr)\n  arr = (arr-meanArr)/stdArr\n  return arr\n\nfor str1 in wineFeatures.columns:\n   wineFeatures[str1] = normalizeData(wineFeatures[str1])\n```\n2. Data Augmentation\n3. NN Pruning\n4. Down sample instead of truncating longer frame sequences\n\nWHY IS OUR VALIDATION SCORE SO BAD\n","metadata":{}},{"cell_type":"markdown","source":"# Imports and Training Params","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tqdm.notebook import tqdm\nfrom multiprocessing import Pool\nimport matplotlib.pyplot as plt\nimport json\nimport zipfile\nimport os\nimport gc\nimport warnings\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n\nBASE_TRAIN_PATH = \"/kaggle/input/asl-signs/\"\nTRAIN_FILE = \"/kaggle/input/asl-signs/train.csv\"\nINDEX_MAP_FILE = '/kaggle/input/asl-signs/sign_to_prediction_index_map.json'\nMODEL_OUT_PATH = '/kaggle/working/modelie'\n\n\nuse_generator = True # Otherwise will load data to disk  \nfind_optimal_params = False\nfilter_outliers = True # Currently unused \ndrop_z = True\nlocal_inference_test = False\n\nepochs = 42\nbatch_size = 64 # Best to have powers of 2\nmax_length = 50 # length that input is padded/truncated to \nLearning_Rate = .001 # Best from keras is .0001\n\nrows_per_frame = 543 #Number of landmarks per frame \nnum_classes = 250 \ndropout_rate = .3\n\nwarnings.filterwarnings(\"ignore\", category=np.VisibleDeprecationWarning) \n\nLEFT_HAND_OFFSET = 468\nPOSE_OFFSET = LEFT_HAND_OFFSET+21\nRIGHT_HAND_OFFSET = POSE_OFFSET+33\nROWS_PER_FRAME = 543\nlip_landmarks = [61, 185, 40, 39, 37,  0, 267, 269, 270, 409, 291,146, 91, 181, 84, 17, 314, 405, 321, 375, 78, 191, 80, 81, 82, 13, 312, 311, 310, 415, 95, 88, 178, 87, 14,317, 402, 318, 324, 308]\n\nleft_hand_landmarks = list(range(LEFT_HAND_OFFSET, LEFT_HAND_OFFSET+21))\nright_hand_landmarks = list(range(RIGHT_HAND_OFFSET, RIGHT_HAND_OFFSET+21))\n\nhand_landmarks = left_hand_landmarks + right_hand_landmarks \n\npoint_landmarks = [item for sublist in [lip_landmarks, left_hand_landmarks, right_hand_landmarks] for item in sublist]\n\ntrain = pd.read_csv(TRAIN_FILE)\nwith open(INDEX_MAP_FILE, 'r') as f: \n    index_map = json.load(f)\n\ntrain['label'] = train['sign'].map(lambda x: index_map[x])\n\nif drop_z: \n    data_columns = ['x', 'y']\nelse: \n    data_columns = ['x', 'y', 'z']","metadata":{"execution":{"iopub.status.busy":"2023-04-11T17:37:39.181943Z","iopub.execute_input":"2023-04-11T17:37:39.182634Z","iopub.status.idle":"2023-04-11T17:37:39.345375Z","shell.execute_reply.started":"2023-04-11T17:37:39.182591Z","shell.execute_reply":"2023-04-11T17:37:39.344297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Process frame path during training","metadata":{}},{"cell_type":"code","source":"def process_frame(path): \n    \n    data = pd.read_parquet(os.path.join(BASE_TRAIN_PATH, path), columns=['x','y','z'])\n        \n    n = int(len(data)/rows_per_frame)\n    \n    data = data.values.reshape(n, rows_per_frame, len(['x', 'y', 'z'])).astype(np.float32)\n    \n    if drop_z:\n        data = data[:, : , :2]\n\n    # Removing the nan hands frames was killing accuracy in inference \n#     nan_hands = list(set(np.where(np.all(np.isnan(data[:, hand_landmarks, :]), axis=1))[0]))\n    \n#     if nan_hands: \n#         data = np.delete(data, nan_hands, axis=0)\n    \n    data = np.nan_to_num(data)\n    \n    data = data[:, point_landmarks, :] # filter for the relevant pose landmarks \n    \n        \n    data = np.reshape(data, (data.shape[0], len(point_landmarks)*len(data_columns)), order = 'C').astype(np.float32)\n    \n    data = data[:max_length, :] # In case we had too many frames \n    \n    if data.shape[0] < max_length: \n        data = np.pad(data, ((0, max_length - data.shape[0]),(0,0)))\n        \n    return data","metadata":{"execution":{"iopub.status.busy":"2023-04-11T17:37:47.305950Z","iopub.execute_input":"2023-04-11T17:37:47.306520Z","iopub.status.idle":"2023-04-11T17:37:47.316639Z","shell.execute_reply.started":"2023-04-11T17:37:47.306481Z","shell.execute_reply":"2023-04-11T17:37:47.315555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Multiprocessed Data Loading (to Disk)\n-Faster","metadata":{}},{"cell_type":"code","source":"if not use_generator: \n    train_paths = [train.iloc[i].path for i in range(int(len(train)))]\n    y = [train.iloc[i].label for i in range(int(len(train)))]\n\n    with Pool(processes=10) as pool: \n        x = list(tqdm(pool.imap(process_frame, train_paths, chunksize=1000), total = 100))\n        pool.close()\n\n    x = np.array(x)\n    y = np.array(y)\n    #x = tf.keras.utils.pad_sequences(x_ragged, padding=\"post\", truncating=\"post\", maxlen = max_length, dtype=np.float32)\n    \n    X_train, X_test, Y_train, Y_test = train_test_split(x, y, test_size=0.2, random_state=1)\n    X_train, X_val, Y_train, Y_val = train_test_split(X_train, Y_train, test_size=0.25, random_state=1) \n\n    del x , y\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-04-11T17:37:53.825544Z","iopub.execute_input":"2023-04-11T17:37:53.825970Z","iopub.status.idle":"2023-04-11T17:37:53.834706Z","shell.execute_reply.started":"2023-04-11T17:37:53.825932Z","shell.execute_reply":"2023-04-11T17:37:53.833492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Generator\n-Allows us to load more data than can fit in memory","metadata":{}},{"cell_type":"code","source":"class DataGenerator(tf.keras.utils.Sequence): \n    def __init__(self, train, list_IDs, point_landmarks, batch_size=32, max_length=30, rows_per_frame=543, \n                 data_columns=['x','y','z'], shuffle=True):\n        self.train = train\n        self.list_IDs = list_IDs\n        self.batch_size = batch_size\n        self.max_length = max_length \n        self.rows_per_frame = rows_per_frame\n        self.point_landmarks = point_landmarks\n        self.data_columns = data_columns \n        self.shuffle = shuffle\n        self.on_epoch_end()\n        \n    def on_epoch_end(self):\n        'Updates indexes after each epoch'\n        self.indexes = np.arange(len(self.list_IDs))\n        if self.shuffle == True:\n            np.random.shuffle(self.indexes)\n    \n    def __data_generation(self, list_IDs_temp):\n        X = np.empty((self.batch_size, self.max_length, len(self.point_landmarks)*len(self.data_columns)))\n        y = np.empty((self.batch_size), dtype=int)\n        \n        for i, ID in enumerate(list_IDs_temp):\n            X[i,] = process_frame(train.iloc[ID].path)\n            y[i] = self.train.iloc[ID].label\n            \n        return X, y\n    \n    def __len__(self):\n        'Denotes the number of batches per epoch'\n        return int(np.floor(len(self.list_IDs) / self.batch_size))\n    \n    def __getitem__(self, index):\n        'Generate one batch of data'\n        # Generate indexes of the batch\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        # Generate data\n        X, y = self.__data_generation(indexes)\n        return X, y\n\nif use_generator: \n    ds_params = {'max_length': max_length, 'batch_size': batch_size, 'rows_per_frame': rows_per_frame,'data_columns': data_columns, 'shuffle': True}\n    partition = {'train': [i for i in range(int(len(train)*.8))], 'validation': [j for j in range(int(len(train)*.8), len(train))]}\n\n    training_generator = DataGenerator(train, partition['train'], point_landmarks, **ds_params)\n    validation_generator = DataGenerator(train, partition['validation'], point_landmarks, **ds_params)","metadata":{"execution":{"iopub.status.busy":"2023-04-11T17:37:57.319940Z","iopub.execute_input":"2023-04-11T17:37:57.320642Z","iopub.status.idle":"2023-04-11T17:37:57.344047Z","shell.execute_reply.started":"2023-04-11T17:37:57.320601Z","shell.execute_reply":"2023-04-11T17:37:57.342944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Frames Stats\nMean: 37.935\nMedian: 22\nStdDev: 44.177\nMax: 537\nMin: 2\n\n### Best Model Params\n160               |units\n\n480               |units_2\n\n416               |units_3\n\n0.0001            |learning_rate             ","metadata":{}},{"cell_type":"code","source":"def get_model(units_1, units_2, units_3): \n    model = tf.keras.Sequential([\n        tf.keras.Input(shape=(max_length, len(point_landmarks)*len(data_columns)), dtype=np.float32), \n        tf.keras.layers.Masking(mask_value=0, input_shape=(max_length, len(point_landmarks)*3)),\n        tf.keras.layers.Dense(units=units_1 , activation='relu'), # Generally want hidden layers to be between the size of the input and output layers \n        tf.keras.layers.Dropout(dropout_rate),\n        tf.keras.layers.LayerNormalization(), \n        tf.keras.layers.Dense(units=units_2, activation='relu'),\n        tf.keras.layers.Dropout(dropout_rate),\n        tf.keras.layers.LayerNormalization(), \n        tf.keras.layers.LSTM(units_3),\n        tf.keras.layers.Dropout(dropout_rate),\n        tf.keras.layers.LayerNormalization(),\n        tf.keras.layers.Flatten(),\n        tf.keras.layers.Dense(units = num_classes, activation='softmax', name='outie'), # Output size is <256>\n    ])\n    model.compile(optimizer = tf.keras.optimizers.Adam(learning_rate=Learning_Rate), loss=\"sparse_categorical_crossentropy\", metrics=[\"accuracy\"])\n    return model \n\nmodel = get_model(160, 480, 416) ","metadata":{"execution":{"iopub.status.busy":"2023-04-11T17:38:03.188940Z","iopub.execute_input":"2023-04-11T17:38:03.189631Z","iopub.status.idle":"2023-04-11T17:38:08.565336Z","shell.execute_reply.started":"2023-04-11T17:38:03.189589Z","shell.execute_reply":"2023-04-11T17:38:08.564266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Optimal Hyper Parameter Search","metadata":{}},{"cell_type":"code","source":"!pip install -q -U keras-tuner\nimport keras_tuner as kt \n\ndef get_tuned_model(hp): \n    hp_units = hp.Int('units', min_value=32, max_value=512, step=32)\n    hp_units_2 = hp.Int('units_2', min_value=32, max_value=512, step=32)\n    hp_units_3 = hp.Int('units_3', min_value=32, max_value=512, step=32)\n    model = get_model(hp_units, hp_units_2, hp_units_3)\n    hp_learning_rate = hp.Choice('learning_rate', values=[1e-2, 1e-3, 1e-4])\n    model.compile(optimizer = tf.keras.optimizers.Adam(learning_rate=hp_learning_rate), \n                  loss=\"sparse_categorical_crossentropy\", metrics=[\"accuracy\"])\n    return model \n\nif find_optimal_params: \n    tuner = kt.Hyperband(get_tuned_model,\n                         objective='val_accuracy',\n                         max_epochs=10,\n                         factor=3,\n                         directory='kaggle/working',\n                         project_name='intro_to_kt')\n\n    stop_early = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=5)\n\n    tuner.search(training_generator, validation_data=validation_generator, epochs=50, callbacks=[stop_early])\n\n    best_hps=tuner.get_best_hyperparameters(num_trials=1)[0]\n    print(best_hps.get('units'))\n    print(best_hps.get('units2'))\n    print(best_hps.get('units3'))\n    print(best_hps.get('learning_rate'))\n    model = tuner.hypermodel.build(best_hps)\n\n    history = model.fit(training_generator, validation_data=validation_generator, epochs=50)\n    val_acc_per_epoch = history.history['val_accuracy']\n    best_epoch = val_acc_per_epoch.index(max(val_acc_per_epoch)) + 1\n    print('Best epoch: %d' % (best_epoch,))\n    epochs = best_epoch","metadata":{"execution":{"iopub.status.busy":"2023-04-11T17:38:15.580163Z","iopub.execute_input":"2023-04-11T17:38:15.580991Z","iopub.status.idle":"2023-04-11T17:38:38.447100Z","shell.execute_reply.started":"2023-04-11T17:38:15.580941Z","shell.execute_reply":"2023-04-11T17:38:38.445551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model fitting using either Data Generator or Numpy Arrays","metadata":{}},{"cell_type":"code","source":"callbacks = [\n        EarlyStopping(\n                monitor = \"val_accuracy\",\n                min_delta = 0, # minimium amount of change to count as an improvement\n                patience = 5, # how many epochs to wait before stopping\n                restore_best_weights=True),\n    \n        ReduceLROnPlateau(monitor = \"val_accuracy\",\n            factor = 0.5,\n            patience = 5)\n            ]\n\nif use_generator: \n    history = model.fit(x=training_generator, \n                        validation_data=validation_generator, \n                        epochs=epochs,\n                        callbacks=callbacks, \n                        use_multiprocessing=True, \n                        workers=6)\n\n    loss, acc = model.evaluate(validation_generator, verbose=2)\nelse: \n    history = model.fit(X_train, Y_train, \n                  batch_size=batch_size, \n                  epochs=epochs,\n                  validation_data = (X_val, Y_val), \n                  callbacks=callbacks)\n\n    loss, acc = model.evaluate(X_val, Y_val, verbose=2)\nprint(\"Restored model, accuracy: {:5.2f}%\".format(100 * acc))","metadata":{"execution":{"iopub.status.busy":"2023-04-11T17:38:53.558270Z","iopub.execute_input":"2023-04-11T17:38:53.558709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Summary\n## Loss and Accuracy Graphs","metadata":{}},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T14:03:47.039867Z","iopub.execute_input":"2023-04-05T14:03:47.040308Z","iopub.status.idle":"2023-04-05T14:03:47.084044Z","shell.execute_reply.started":"2023-04-05T14:03:47.040278Z","shell.execute_reply":"2023-04-05T14:03:47.082709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T14:03:49.846278Z","iopub.execute_input":"2023-04-05T14:03:49.846642Z","iopub.status.idle":"2023-04-05T14:03:50.071634Z","shell.execute_reply.started":"2023-04-05T14:03:49.846615Z","shell.execute_reply":"2023-04-05T14:03:50.069600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T14:03:53.139171Z","iopub.execute_input":"2023-04-05T14:03:53.139547Z","iopub.status.idle":"2023-04-05T14:03:53.297243Z","shell.execute_reply.started":"2023-04-05T14:03:53.139519Z","shell.execute_reply":"2023-04-05T14:03:53.296121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference Model","metadata":{}},{"cell_type":"code","source":"class FeatureGenTF(tf.keras.layers.Layer):\n    def __init__(self):\n        super().__init__()\n        \n\n    def call(self, x):  \n        if drop_z: \n            x = x[:, : , :-1]\n\n        x = tf.gather(x, point_landmarks, axis=1)\n        \n        # Removing nan hand frames was killing accuracy for some reason \n#         not_nan_frames = tf.where(tf.logical_not(tf.reduce_all(tf.reduce_any(tf.math.is_nan(x[:, 40:, :]), axis = 2), axis=1)))[0]\n        \n#         x = tf.gather(x, not_nan_frames, axis=0) \n                \n        x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n        \n        x = tf.image.resize_with_pad(x, max_length, len(point_landmarks))  \n                \n        x = tf.reshape(x, (max_length, len(point_landmarks)*len(data_columns)))\n        \n        x = tf.expand_dims(x,0)\n        \n        return x\n    \npreprocessing = FeatureGenTF()","metadata":{"execution":{"iopub.status.busy":"2023-04-05T16:42:38.346597Z","iopub.execute_input":"2023-04-05T16:42:38.348461Z","iopub.status.idle":"2023-04-05T16:42:38.360544Z","shell.execute_reply.started":"2023-04-05T16:42:38.348339Z","shell.execute_reply":"2023-04-05T16:42:38.358649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_inference_model(model):\n    inputs = tf.keras.Input((543, 3), dtype=tf.float32, name=\"inputs\") \n    x = preprocessing(inputs)\n    x = model(x)\n    output = tf.keras.layers.Activation(activation=\"linear\", name=\"outputs\")(x)\n    inference_model = tf.keras.Model(inputs=inputs, outputs=output) \n    inference_model.compile(loss=tf.keras.losses.SparseCategoricalCrossentropy(), metrics=[\"accuracy\"])\n    return inference_model\n\ninference_model = get_inference_model(model)\ninference_model.summary(expand_nested=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-05T16:42:40.683567Z","iopub.execute_input":"2023-04-05T16:42:40.683939Z","iopub.status.idle":"2023-04-05T16:42:41.663793Z","shell.execute_reply.started":"2023-04-05T16:42:40.683910Z","shell.execute_reply":"2023-04-05T16:42:41.661526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"converter = tf.lite.TFLiteConverter.from_keras_model(inference_model)\ntflite_model = converter.convert()\n\nmodel_path = \"model.tflite\"\n\nwith open(model_path, 'wb') as f:\n    f.write(tflite_model)\n\n!zip submission.zip $model_path","metadata":{"execution":{"iopub.status.busy":"2023-04-05T16:42:43.910419Z","iopub.execute_input":"2023-04-05T16:42:43.910793Z","iopub.status.idle":"2023-04-05T16:43:04.577489Z","shell.execute_reply.started":"2023-04-05T16:42:43.910764Z","shell.execute_reply":"2023-04-05T16:43:04.576114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Testing inference model","metadata":{}},{"cell_type":"code","source":"!pip install tflite_runtime\nimport tflite_runtime.interpreter as tflite\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)\n\nif local_inference_test: \n    #TODO Find out what signs the model is bad at inferring\n    interpreter = tflite.Interpreter(model_path)\n    found_signatures = list(interpreter.get_signature_list().keys())\n    prediction_fn = interpreter.get_signature_runner(\"serving_default\")\n    p2s_map = {v: k for k, v in index_map.items()}\n    decoder = lambda x: p2s_map.get(x)\n    score = 0 \n    for i in tqdm(range(int(len(train)/10))): \n        frames = load_relevant_data_subset(os.path.join(BASE_TRAIN_PATH, train.iloc[i].path))\n        output = prediction_fn(inputs=frames)\n        sign = np.argmax(output[\"outputs\"])\n        if decoder(train.iloc[i].label) == decoder(sign): \n            score += 1 \n    print(score/int(len(train)/10))","metadata":{"execution":{"iopub.status.busy":"2023-04-05T16:24:00.486862Z","iopub.execute_input":"2023-04-05T16:24:00.487271Z","iopub.status.idle":"2023-04-05T16:26:34.493313Z","shell.execute_reply.started":"2023-04-05T16:24:00.487236Z","shell.execute_reply":"2023-04-05T16:26:34.491915Z"},"trusted":true},"execution_count":null,"outputs":[]}]}