{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-28T09:31:12.520412Z","iopub.execute_input":"2023-03-28T09:31:12.522176Z","iopub.status.idle":"2023-03-28T09:31:12.530733Z","shell.execute_reply.started":"2023-03-28T09:31:12.522103Z","shell.execute_reply":"2023-03-28T09:31:12.528834Z"},"trusted":true},"execution_count":13,"outputs":[]},{"cell_type":"code","source":"import os\nimport tensorflow as tf\nimport numpy as np\nimport pandas as pd\nimport json\nfrom tqdm import tqdm\nimport tensorflow as tf\n\nQUICK_TEST = False\nQUICK_LIMIT = 200\n\nos.environ.update({\n    'TF_DETERMINISTIC_OPS': '1',\n    'TF_CUDNN_DETERMINISTIC': '1'\n})\n\n# print(tf.config.list_physical_devices('GPU'))\n\n# physical_devices = tf.config.list_physical_devices('GPU')\n# tf.config.experimental.set_memory_growth(physical_devices[0], True)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T21:48:12.084494Z","iopub.execute_input":"2023-03-28T21:48:12.084908Z","iopub.status.idle":"2023-03-28T21:48:12.092658Z","shell.execute_reply.started":"2023-03-28T21:48:12.084871Z","shell.execute_reply":"2023-03-28T21:48:12.091003Z"},"trusted":true},"execution_count":4,"outputs":[]},{"cell_type":"code","source":"# Force TensorFlow to use the GPU\n# gpus = tf.config.experimental.list_physical_devices('GPU')\n# if gpus:\n#     try:\n#         tf.config.experimental.set_visible_devices(gpus[0], 'GPU')\n#         tf.config.experimental.set_memory_growth(gpus[0], True)\n#         print(\"GPU is being used...\")\n#     except RuntimeError as e:\n#         print(e)\n# else:\n#     print(\"GPU is not available\")\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T09:31:12.550907Z","iopub.execute_input":"2023-03-28T09:31:12.552238Z","iopub.status.idle":"2023-03-28T09:31:12.559387Z","shell.execute_reply.started":"2023-03-28T09:31:12.552193Z","shell.execute_reply":"2023-03-28T09:31:12.558494Z"},"trusted":true},"execution_count":15,"outputs":[]},{"cell_type":"code","source":"\n# print(\"\\n... PIP INSTALLS STARTING ...\\n\")\n# !pip install -q --upgrade tensorflow-io\n# try:\n#     import mediapipe as mp\n# except ModuleNotFoundError:\n#     !pip install -q mediapipe\n#     import mediapipe as mp\n# print(\"\\n... PIP INSTALLS COMPLETE ...\\n\")\n\n# !pip install -q lion-tf\n\n# print(\"\\n... IMPORTS STARTING ...\\n\")\n# print(\"\\n\\tVERSION INFORMATION\")\n\n# # Machine Learning and Data Science Imports\n# import pandas as pd\n# import numpy as np\n# import sklearn\n# import tensorflow as tf\n# import tensorflow_io as tfio\n\n# print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\")\n# print(f\"\\t\\t– TENSORFLOW-IO VERSION: {tfio.__version__}\")\n# print(f\"\\t\\t– NUMPY VERSION: {np.__version__}\")\n# print(f\"\\t\\t– SKLEARN VERSION: {sklearn.__version__}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-28T09:31:12.560786Z","iopub.execute_input":"2023-03-28T09:31:12.561194Z","iopub.status.idle":"2023-03-28T09:31:12.571959Z","shell.execute_reply.started":"2023-03-28T09:31:12.561159Z","shell.execute_reply":"2023-03-28T09:31:12.570797Z"},"trusted":true},"execution_count":16,"outputs":[]},{"cell_type":"code","source":"# !mkdir -p /kaggle/working/gislr-feature-data-on-the-shoulders/\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T09:31:12.574046Z","iopub.execute_input":"2023-03-28T09:31:12.57504Z","iopub.status.idle":"2023-03-28T09:31:12.589576Z","shell.execute_reply.started":"2023-03-28T09:31:12.574995Z","shell.execute_reply":"2023-03-28T09:31:12.587978Z"},"trusted":true},"execution_count":17,"outputs":[]},{"cell_type":"code","source":"# Built-In Imports\nfrom kaggle_datasets import KaggleDatasets\nfrom collections import Counter\nfrom datetime import datetime\nfrom zipfile import ZipFile\nfrom glob import glob\nimport Levenshtein\nimport warnings\nimport requests\nimport hashlib\nimport random\nimport json","metadata":{"execution":{"iopub.status.busy":"2023-03-28T21:48:15.932195Z","iopub.execute_input":"2023-03-28T21:48:15.932634Z","iopub.status.idle":"2023-03-28T21:48:16.037256Z","shell.execute_reply.started":"2023-03-28T21:48:15.93259Z","shell.execute_reply":"2023-03-28T21:48:16.035688Z"},"trusted":true},"execution_count":5,"outputs":[]},{"cell_type":"code","source":"\n\ndef seed_it_all(seed=7):\n    \"\"\" Attempt to be Reproducible \"\"\"\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\nseed_it_all()\n\nprint(\"\\n\\n... IMPORTS COMPLETE ...\\n\")\n\n# Define a function to load the JSON data from a file\ndef read_json_file(file_path):\n    try:\n        with open(file_path, 'r') as file:\n            json_data = json.load(file)\n        return json_data\n    except FileNotFoundError:\n        raise FileNotFoundError(f\"File not found: {file_path}\")\n    except ValueError:\n        raise ValueError(f\"Invalid JSON data in file: {file_path}\")\n\n# Define the path to the root data directory\nDATA_DIR = \"/kaggle/input/asl-signs\"\n\nprint(\"\\n... BASIC DATA SETUP STARTING ...\\n\")\n\nprint(\"\\n\\n... from CSV file load train dataframe  ...\\n\")\ntrain_df = pd.read_csv(os.path.join(DATA_DIR, \"train.csv\"))\ntrain_df[\"path\"] = DATA_DIR + \"/\" + train_df[\"path\"]\ndisplay(train_df)\n\nprint(\"\\n\\n... load sign_to_prediction_index_map-json ...\\n\")\nsign_map = {k.lower(): v for k, v in read_json_file(os.path.join(DATA_DIR, \"sign_to_prediction_index_map.json\")).items()}\nprint(sign_map)","metadata":{"execution":{"iopub.status.busy":"2023-03-28T21:48:18.438363Z","iopub.execute_input":"2023-03-28T21:48:18.438834Z","iopub.status.idle":"2023-03-28T21:48:18.752738Z","shell.execute_reply.started":"2023-03-28T21:48:18.438791Z","shell.execute_reply":"2023-03-28T21:48:18.751332Z"},"trusted":true},"execution_count":6,"outputs":[{"name":"stdout","text":"\n\n... IMPORTS COMPLETE ...\n\n\n... BASIC DATA SETUP STARTING ...\n\n\n\n... from CSV file load train dataframe  ...\n\n","output_type":"stream"},{"output_type":"display_data","data":{"text/plain":"                                                    path  participant_id  \\\n0      /kaggle/input/asl-signs/train_landmark_files/2...           26734   \n1      /kaggle/input/asl-signs/train_landmark_files/2...           28656   \n2      /kaggle/input/asl-signs/train_landmark_files/1...           16069   \n3      /kaggle/input/asl-signs/train_landmark_files/2...           25571   \n4      /kaggle/input/asl-signs/train_landmark_files/6...           62590   \n...                                                  ...             ...   \n94472  /kaggle/input/asl-signs/train_landmark_files/5...           53618   \n94473  /kaggle/input/asl-signs/train_landmark_files/2...           26734   \n94474  /kaggle/input/asl-signs/train_landmark_files/2...           25571   \n94475  /kaggle/input/asl-signs/train_landmark_files/2...           29302   \n94476  /kaggle/input/asl-signs/train_landmark_files/3...           36257   \n\n       sequence_id    sign  \n0       1000035562    blow  \n1       1000106739    wait  \n2        100015657   cloud  \n3       1000210073    bird  \n4       1000240708    owie  \n...            ...     ...  \n94472    999786174   white  \n94473    999799849    have  \n94474    999833418  flower  \n94475    999895257    room  \n94476    999962374   happy  \n\n[94477 rows x 4 columns]","text/html":"<div>\n<style scoped>\n    .dataframe tbody tr th:only-of-type {\n        vertical-align: middle;\n    }\n\n    .dataframe tbody tr th {\n        vertical-align: top;\n    }\n\n    .dataframe thead th {\n        text-align: right;\n    }\n</style>\n<table border=\"1\" class=\"dataframe\">\n  <thead>\n    <tr style=\"text-align: right;\">\n      <th></th>\n      <th>path</th>\n      <th>participant_id</th>\n      <th>sequence_id</th>\n      <th>sign</th>\n    </tr>\n  </thead>\n  <tbody>\n    <tr>\n      <th>0</th>\n      <td>/kaggle/input/asl-signs/train_landmark_files/2...</td>\n      <td>26734</td>\n      <td>1000035562</td>\n      <td>blow</td>\n    </tr>\n    <tr>\n      <th>1</th>\n      <td>/kaggle/input/asl-signs/train_landmark_files/2...</td>\n      <td>28656</td>\n      <td>1000106739</td>\n      <td>wait</td>\n    </tr>\n    <tr>\n      <th>2</th>\n      <td>/kaggle/input/asl-signs/train_landmark_files/1...</td>\n      <td>16069</td>\n      <td>100015657</td>\n      <td>cloud</td>\n    </tr>\n    <tr>\n      <th>3</th>\n      <td>/kaggle/input/asl-signs/train_landmark_files/2...</td>\n      <td>25571</td>\n      <td>1000210073</td>\n      <td>bird</td>\n    </tr>\n    <tr>\n      <th>4</th>\n      <td>/kaggle/input/asl-signs/train_landmark_files/6...</td>\n      <td>62590</td>\n      <td>1000240708</td>\n      <td>owie</td>\n    </tr>\n    <tr>\n      <th>...</th>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n      <td>...</td>\n    </tr>\n    <tr>\n      <th>94472</th>\n      <td>/kaggle/input/asl-signs/train_landmark_files/5...</td>\n      <td>53618</td>\n      <td>999786174</td>\n      <td>white</td>\n    </tr>\n    <tr>\n      <th>94473</th>\n      <td>/kaggle/input/asl-signs/train_landmark_files/2...</td>\n      <td>26734</td>\n      <td>999799849</td>\n      <td>have</td>\n    </tr>\n    <tr>\n      <th>94474</th>\n      <td>/kaggle/input/asl-signs/train_landmark_files/2...</td>\n      <td>25571</td>\n      <td>999833418</td>\n      <td>flower</td>\n    </tr>\n    <tr>\n      <th>94475</th>\n      <td>/kaggle/input/asl-signs/train_landmark_files/2...</td>\n      <td>29302</td>\n      <td>999895257</td>\n      <td>room</td>\n    </tr>\n    <tr>\n      <th>94476</th>\n      <td>/kaggle/input/asl-signs/train_landmark_files/3...</td>\n      <td>36257</td>\n      <td>999962374</td>\n      <td>happy</td>\n    </tr>\n  </tbody>\n</table>\n<p>94477 rows × 4 columns</p>\n</div>"},"metadata":{}},{"name":"stdout","text":"\n\n... load sign_to_prediction_index_map-json ...\n\n{'tv': 0, 'after': 1, 'airplane': 2, 'all': 3, 'alligator': 4, 'animal': 5, 'another': 6, 'any': 7, 'apple': 8, 'arm': 9, 'aunt': 10, 'awake': 11, 'backyard': 12, 'bad': 13, 'balloon': 14, 'bath': 15, 'because': 16, 'bed': 17, 'bedroom': 18, 'bee': 19, 'before': 20, 'beside': 21, 'better': 22, 'bird': 23, 'black': 24, 'blow': 25, 'blue': 26, 'boat': 27, 'book': 28, 'boy': 29, 'brother': 30, 'brown': 31, 'bug': 32, 'bye': 33, 'callonphone': 34, 'can': 35, 'car': 36, 'carrot': 37, 'cat': 38, 'cereal': 39, 'chair': 40, 'cheek': 41, 'child': 42, 'chin': 43, 'chocolate': 44, 'clean': 45, 'close': 46, 'closet': 47, 'cloud': 48, 'clown': 49, 'cow': 50, 'cowboy': 51, 'cry': 52, 'cut': 53, 'cute': 54, 'dad': 55, 'dance': 56, 'dirty': 57, 'dog': 58, 'doll': 59, 'donkey': 60, 'down': 61, 'drawer': 62, 'drink': 63, 'drop': 64, 'dry': 65, 'dryer': 66, 'duck': 67, 'ear': 68, 'elephant': 69, 'empty': 70, 'every': 71, 'eye': 72, 'face': 73, 'fall': 74, 'farm': 75, 'fast': 76, 'feet': 77, 'find': 78, 'fine': 79, 'finger': 80, 'finish': 81, 'fireman': 82, 'first': 83, 'fish': 84, 'flag': 85, 'flower': 86, 'food': 87, 'for': 88, 'frenchfries': 89, 'frog': 90, 'garbage': 91, 'gift': 92, 'giraffe': 93, 'girl': 94, 'give': 95, 'glasswindow': 96, 'go': 97, 'goose': 98, 'grandma': 99, 'grandpa': 100, 'grass': 101, 'green': 102, 'gum': 103, 'hair': 104, 'happy': 105, 'hat': 106, 'hate': 107, 'have': 108, 'haveto': 109, 'head': 110, 'hear': 111, 'helicopter': 112, 'hello': 113, 'hen': 114, 'hesheit': 115, 'hide': 116, 'high': 117, 'home': 118, 'horse': 119, 'hot': 120, 'hungry': 121, 'icecream': 122, 'if': 123, 'into': 124, 'jacket': 125, 'jeans': 126, 'jump': 127, 'kiss': 128, 'kitty': 129, 'lamp': 130, 'later': 131, 'like': 132, 'lion': 133, 'lips': 134, 'listen': 135, 'look': 136, 'loud': 137, 'mad': 138, 'make': 139, 'man': 140, 'many': 141, 'milk': 142, 'minemy': 143, 'mitten': 144, 'mom': 145, 'moon': 146, 'morning': 147, 'mouse': 148, 'mouth': 149, 'nap': 150, 'napkin': 151, 'night': 152, 'no': 153, 'noisy': 154, 'nose': 155, 'not': 156, 'now': 157, 'nuts': 158, 'old': 159, 'on': 160, 'open': 161, 'orange': 162, 'outside': 163, 'owie': 164, 'owl': 165, 'pajamas': 166, 'pen': 167, 'pencil': 168, 'penny': 169, 'person': 170, 'pig': 171, 'pizza': 172, 'please': 173, 'police': 174, 'pool': 175, 'potty': 176, 'pretend': 177, 'pretty': 178, 'puppy': 179, 'puzzle': 180, 'quiet': 181, 'radio': 182, 'rain': 183, 'read': 184, 'red': 185, 'refrigerator': 186, 'ride': 187, 'room': 188, 'sad': 189, 'same': 190, 'say': 191, 'scissors': 192, 'see': 193, 'shhh': 194, 'shirt': 195, 'shoe': 196, 'shower': 197, 'sick': 198, 'sleep': 199, 'sleepy': 200, 'smile': 201, 'snack': 202, 'snow': 203, 'stairs': 204, 'stay': 205, 'sticky': 206, 'store': 207, 'story': 208, 'stuck': 209, 'sun': 210, 'table': 211, 'talk': 212, 'taste': 213, 'thankyou': 214, 'that': 215, 'there': 216, 'think': 217, 'thirsty': 218, 'tiger': 219, 'time': 220, 'tomorrow': 221, 'tongue': 222, 'tooth': 223, 'toothbrush': 224, 'touch': 225, 'toy': 226, 'tree': 227, 'uncle': 228, 'underwear': 229, 'up': 230, 'vacuum': 231, 'wait': 232, 'wake': 233, 'water': 234, 'wet': 235, 'weus': 236, 'where': 237, 'white': 238, 'who': 239, 'why': 240, 'will': 241, 'wolf': 242, 'yellow': 243, 'yes': 244, 'yesterday': 245, 'yourself': 246, 'yucky': 247, 'zebra': 248, 'zipper': 249}\n","output_type":"stream"}]},{"cell_type":"code","source":"# Define the preprocessing configuration\npreprocessing_config = {\n    'drop_z': False,\n    'num_frames': 15,\n    'segments': 3,\n    'left_hand_offset': 468,\n    'pose_offset': 489,\n    'right_hand_offset': 522,\n    'averaging_sets': [[0, 468], [489, 33]],\n    'lip_landmarks': [61, 185, 40, 39, 37,  0, 267, 269, 270, 409, 291, 146, 91, 181, 84, 17, 314, 405, 321, 375, \n                      78, 191, 80, 81, 82, 13, 312, 311, 310, 415, 95, 88, 178, 87, 14, 317, 402, 318, 324, 308],\n    'left_hand_landmarks': list(range(468, 489)),\n    'right_hand_landmarks': list(range(522, 543)),\n    'point_landmarks': [],\n    'landmarks': 0,\n    'input_shape': (),\n    'flat_input_shape': ()\n}\n\n# Compute the point landmarks\npreprocessing_config['point_landmarks'] = [item for sublist in \n                                           [preprocessing_config['lip_landmarks'], \n                                            preprocessing_config['left_hand_landmarks'], \n                                            preprocessing_config['right_hand_landmarks']] \n                                           for item in sublist]\n\n# Compute the number of landmarks\npreprocessing_config['landmarks'] = len(preprocessing_config['point_landmarks']) + len(preprocessing_config['averaging_sets'])\n\n# Compute the input shape\nif preprocessing_config['drop_z']:\n    preprocessing_config['input_shape'] = (preprocessing_config['num_frames'], preprocessing_config['landmarks']*2)\nelse:\n    preprocessing_config['input_shape'] = (preprocessing_config['num_frames'], preprocessing_config['landmarks']*3)\n\n# Compute the flat input shape\npreprocessing_config['flat_input_shape'] = (preprocessing_config['input_shape'][0] + 2 * (preprocessing_config['segments'] + 1)) * preprocessing_config['input_shape'][1]\n\nprint(preprocessing_config['landmarks'])\n\nDROP_Z = preprocessing_config['drop_z']\n\nFLAT_INPUT_SHAPE = (preprocessing_config['input_shape'][0] + 2 * (preprocessing_config['segments'] + 1)) * preprocessing_config['input_shape'][1]\n\naveraging_sets = preprocessing_config['averaging_sets']\npoint_landmarks = preprocessing_config['point_landmarks']\nSEGMENTS = preprocessing_config['segments']\nINPUT_SHAPE = preprocessing_config['input_shape']\nNUM_FRAMES = preprocessing_config['num_frames']\nLANDMARKS = preprocessing_config['landmarks']","metadata":{"execution":{"iopub.status.busy":"2023-03-28T21:48:22.319515Z","iopub.execute_input":"2023-03-28T21:48:22.319928Z","iopub.status.idle":"2023-03-28T21:48:22.33868Z","shell.execute_reply.started":"2023-03-28T21:48:22.319894Z","shell.execute_reply":"2023-03-28T21:48:22.337178Z"},"trusted":true},"execution_count":7,"outputs":[{"name":"stdout","text":"84\n","output_type":"stream"}]},{"cell_type":"code","source":"# Define constants\nLANDMARK_FILES_DIR = \"/kaggle/input/asl-signs/train_landmark_files\"\nTRAIN_FILE = \"/kaggle/input/asl-signs/train.csv\"\nlabel_map = json.load(open(\"/kaggle/input/asl-signs/sign_to_prediction_index_map.json\", \"r\"))","metadata":{"execution":{"iopub.status.busy":"2023-03-28T21:48:25.802243Z","iopub.execute_input":"2023-03-28T21:48:25.802646Z","iopub.status.idle":"2023-03-28T21:48:25.810114Z","shell.execute_reply.started":"2023-03-28T21:48:25.802611Z","shell.execute_reply":"2023-03-28T21:48:25.808322Z"},"trusted":true},"execution_count":8,"outputs":[]},{"cell_type":"code","source":"ROWS_PER_FRAME = 543  # number of landmarks per frame\ndef load_relevant_data_landmarks(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    \n    # Remove NaN values\n    data = np.nan_to_num(data, nan=0.0)\n    \n    return data.astype(np.float32)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T21:48:27.626497Z","iopub.execute_input":"2023-03-28T21:48:27.626911Z","iopub.status.idle":"2023-03-28T21:48:27.634018Z","shell.execute_reply.started":"2023-03-28T21:48:27.626877Z","shell.execute_reply":"2023-03-28T21:48:27.633047Z"},"trusted":true},"execution_count":9,"outputs":[]},{"cell_type":"code","source":"# Helper Functions\ndef tf_nan_mean(x, axis=0):\n    return tf.reduce_sum(tf.debugging.check_numerics(x, \"NaN values found\"), axis=axis) / tf.reduce_sum(tf.debugging.check_numerics(tf.ones_like(x), \"NaN values found\"), axis=axis)\n\ndef tf_nan_std(x, axis=0):\n    d = x - tf_nan_mean(x, axis=axis)\n    return tf.math.sqrt(tf_nan_mean(d * d, axis=axis))\n\ndef flatten_means_and_stds(x, axis=0):\n    # Get means and stds\n    x_mean = tf_nan_mean(x, axis=0)\n    x_std  = tf_nan_std(x,  axis=0)\n\n    x_out = tf.concat([x_mean, x_std], axis=0)\n    x_out = tf.reshape(x_out, (1, INPUT_SHAPE[1]*2))\n    x_out = tf.debugging.check_numerics(x_out, \"NaN values found\")\n    return x_out\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T21:48:30.850732Z","iopub.execute_input":"2023-03-28T21:48:30.851299Z","iopub.status.idle":"2023-03-28T21:48:30.863933Z","shell.execute_reply.started":"2023-03-28T21:48:30.851244Z","shell.execute_reply":"2023-03-28T21:48:30.862543Z"},"trusted":true},"execution_count":10,"outputs":[]},{"cell_type":"code","source":"class FeatureGen(tf.keras.layers.Layer):\n    def __init__(self):\n        super(FeatureGen, self).__init__()\n\n    def call(self, x_in):\n        if DROP_Z:\n            x_in = x_in[:, :, 0:2]\n        x_list = [tf.expand_dims(tf.math.reduce_mean(tf.where(tf.math.is_finite(x_in[:, av_set[0]:av_set[0]+av_set[1], :]), x_in[:, av_set[0]:av_set[0]+av_set[1], :], 0), axis=1), axis=1) for av_set in averaging_sets]\n        x_list.append(tf.gather(x_in, point_landmarks, axis=1))\n        x = tf.concat(x_list, 1)\n\n        # Pad the tensor along the first dimension\n        pad_len = SEGMENTS - (tf.shape(x)[0] % SEGMENTS)\n        x = tf.pad(x, [[0, pad_len], [0, 0], [0, 0]], mode=\"SYMMETRIC\")\n\n        # Split the padded tensor into segments\n        x_list = tf.split(x, SEGMENTS)\n\n        # Apply flatten_means_and_stds to each segment\n        x_list = [flatten_means_and_stds(_x, axis=0) for _x in x_list]\n\n        x_list.append(flatten_means_and_stds(x, axis=0))\n\n        # Resize and flatten the final tensor\n        x = tf.image.resize(tf.where(tf.math.is_finite(x), x, tf.math.reduce_mean(x, axis=0, keepdims=True)), [NUM_FRAMES, LANDMARKS])\n        x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n        x_list.append(tf.reshape(x, (1, -1)))\n        x = tf.concat(x_list, axis=1)\n        return x\n\n\n\nfeature_converter = FeatureGen()\nprint(feature_converter(tf.keras.Input((543, 3), dtype=tf.float32, name=\"inputs\")))\nfeature_converter(load_relevant_data_landmarks(f'/kaggle/input/asl-signs/{pd.read_csv(TRAIN_FILE).path[1]}'))\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T21:48:34.634404Z","iopub.execute_input":"2023-03-28T21:48:34.63488Z","iopub.status.idle":"2023-03-28T21:48:36.293604Z","shell.execute_reply.started":"2023-03-28T21:48:34.634834Z","shell.execute_reply":"2023-03-28T21:48:36.292333Z"},"trusted":true},"execution_count":11,"outputs":[{"name":"stdout","text":"KerasTensor(type_spec=TensorSpec(shape=(1, 5796), dtype=tf.float32, name=None), name='feature_gen/concat_5:0', description=\"created by layer 'feature_gen'\")\n","output_type":"stream"},{"execution_count":11,"output_type":"execute_result","data":{"text/plain":"<tf.Tensor: shape=(1, 5796), dtype=float32, numpy=\narray([[ 5.7140142e-01,  4.6732509e-01, -2.4906856e-05, ...,\n         3.8143307e-01,  8.0332047e-01, -1.0697230e-01]], dtype=float32)>"},"metadata":{}}]},{"cell_type":"code","source":"\n# print(FeatureGen(tf.keras.Input((543, 3), dtype=tf.float32, name=\"inputs\")))\n# FeatureGen(drop_z=True)(load_relevant_data_landmarks(train_df.path[0]))","metadata":{"execution":{"iopub.status.busy":"2023-03-28T09:31:13.276813Z","iopub.execute_input":"2023-03-28T09:31:13.277419Z","iopub.status.idle":"2023-03-28T09:31:13.284232Z","shell.execute_reply.started":"2023-03-28T09:31:13.277375Z","shell.execute_reply":"2023-03-28T09:31:13.282134Z"},"trusted":true},"execution_count":25,"outputs":[]},{"cell_type":"code","source":"def convert_row(row, right_handed=True):\n    x = load_relevant_data_landmarks(os.path.join(\"/kaggle/input/asl-signs\", row[1].path))\n    x = feature_converter(tf.convert_to_tensor(x)).cpu().numpy()\n    return x, row[1].label\n\n# right_handed_signer = [26734, 28656, 25571, 62590, 29302, \n#                        49445, 53618, 18796,  4718,  2044, \n#                        37779, 30680]\n# left_handed_signer  = [16069, 32319, 36257, 22343, 27610, \n#                        61333, 34503, 55372, ]\n# both_hands_signer   = [37055, ]\n\n# messy = [29302, ]\n\ndef convert_and_save_data():\n    df = pd.read_csv(TRAIN_FILE)\n    df['label'] = df['sign'].map(label_map)\n    total = df.shape[0]\n    if QUICK_TEST:\n        total = QUICK_LIMIT\n    npdata = np.zeros((total, INPUT_SHAPE[0]*INPUT_SHAPE[1] + (SEGMENTS+1)*INPUT_SHAPE[1]*2))\n    nplabels = np.zeros(total)\n    for i, row in tqdm(enumerate(df.iterrows()), total=total):\n        (x,y) = convert_row(row)\n        npdata[i,:] = x\n        nplabels[i] = y\n        if QUICK_TEST and i == QUICK_LIMIT - 1:\n            break\n    \n    np.save(\"feature_data.npy\", npdata)\n    np.save(\"feature_labels.npy\", nplabels)\n        \nconvert_and_save_data()\n\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T09:31:13.288773Z","iopub.execute_input":"2023-03-28T09:31:13.289254Z","iopub.status.idle":"2023-03-28T10:29:07.896643Z","shell.execute_reply.started":"2023-03-28T09:31:13.289217Z","shell.execute_reply":"2023-03-28T10:29:07.893682Z"},"trusted":true},"execution_count":26,"outputs":[{"name":"stderr","text":"100%|██████████| 94477/94477 [57:43<00:00, 27.28it/s]  \n","output_type":"stream"}]},{"cell_type":"code","source":"# Load the saved data and labels\nX = np.load(\"feature_data.npy\")\ny = np.load(\"feature_labels.npy\")\n\n# Print the shapes of the data and labels arrays\nprint(X.shape, y.shape)\n\n# Print an example row of the data array\nprint(X[0,:].shape, X[0,:])\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T21:49:21.411901Z","iopub.execute_input":"2023-03-28T21:49:21.412314Z","iopub.status.idle":"2023-03-28T21:49:21.451333Z","shell.execute_reply.started":"2023-03-28T21:49:21.412276Z","shell.execute_reply":"2023-03-28T21:49:21.449241Z"},"trusted":true},"execution_count":14,"outputs":[{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mFileNotFoundError\u001b[0m                         Traceback (most recent call last)","\u001b[0;32m/tmp/ipykernel_27/2851289613.py\u001b[0m in \u001b[0;36m<module>\u001b[0;34m\u001b[0m\n\u001b[1;32m      1\u001b[0m \u001b[0;31m# Load the saved data and labels\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m----> 2\u001b[0;31m \u001b[0mX\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mnp\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mload\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m\"feature_data.npy\"\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m      3\u001b[0m \u001b[0my\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mnp\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mload\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m\"feature_labels.npy\"\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m      4\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m      5\u001b[0m \u001b[0;31m# Print the shapes of the data and labels arrays\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/opt/conda/lib/python3.7/site-packages/numpy/lib/npyio.py\u001b[0m in \u001b[0;36mload\u001b[0;34m(file, mmap_mode, allow_pickle, fix_imports, encoding)\u001b[0m\n\u001b[1;32m    415\u001b[0m             \u001b[0mown_fid\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0;32mFalse\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    416\u001b[0m         \u001b[0;32melse\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m--> 417\u001b[0;31m             \u001b[0mfid\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mstack\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0menter_context\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mopen\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mos_fspath\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mfile\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m\"rb\"\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m    418\u001b[0m             \u001b[0mown_fid\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0;32mTrue\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m    419\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;31mFileNotFoundError\u001b[0m: [Errno 2] No such file or directory: 'feature_data.npy'"],"ename":"FileNotFoundError","evalue":"[Errno 2] No such file or directory: 'feature_data.npy'","output_type":"error"}]},{"cell_type":"code","source":"import tensorflow as tf\n\n# Define hyperparameters\nBATCH_SIZE = 128\nVAL_PCT = 0.1\nLEARNING_RATE = 0.000343\nLR_PATIENCE = 2\nLR_REDUCTION_FACTOR = 0.8\nEPOCHS = 50\n\nSTARTING_LAYER_SIZE = 1024\nDROPOUTS = [0.4, 0.5]\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:29:16.571385Z","iopub.execute_input":"2023-03-28T10:29:16.571781Z","iopub.status.idle":"2023-03-28T10:29:16.579648Z","shell.execute_reply.started":"2023-03-28T10:29:16.571744Z","shell.execute_reply":"2023-03-28T10:29:16.578147Z"},"trusted":true},"execution_count":28,"outputs":[]},{"cell_type":"code","source":"\n# Load data\ntrain_x = np.load(\"/kaggle/working/feature_data.npy\").astype(np.float32)\ntrain_y = np.load(\"/kaggle/working/feature_labels.npy\").astype(np.uint8)\n\n# Preprocess data\nif DROP_Z:\n    train_x = np.reshape(train_x, [train_x.shape[0], -1, 3])\n    train_x = train_x[:, :, 0:2]\n    train_x = np.reshape(train_x, [train_x.shape[0], -1])\n\nN_TOTAL = train_x.shape[0]\nFLAT_FRAME_SHAPE = train_x.shape[1]\nassert(FLAT_FRAME_SHAPE == FLAT_INPUT_SHAPE)\nN_VAL   = int(N_TOTAL*VAL_PCT)\nN_TRAIN = N_TOTAL-N_VAL\n\nrandom_idxs = random.sample(range(N_TOTAL), N_TOTAL)\ntrain_idxs, val_idxs = np.array(random_idxs[:N_TRAIN], dtype=np.int32), np.array(random_idxs[N_TRAIN:], dtype=np.int32)\nprint(\"train_idxs:\", type(train_idxs))\nprint(\"val_idxs:\", type(val_idxs))\n\nval_x, val_y = train_x[val_idxs], train_y[val_idxs]\ntrain_x, train_y = train_x[train_idxs], train_y[train_idxs]\n\nprint(\"val_x\", val_x.shape)\nprint(\"val_y\", val_y.shape)\nprint(\"train_x\", train_x.shape)\nprint(\"train_y\", train_y.shape)\n# Create dataset\ntrain_ds = tf.data.Dataset.from_tensor_slices((train_x, train_y)).shuffle(10000).batch(BATCH_SIZE)\nval_ds = tf.data.Dataset.from_tensor_slices((val_x, val_y)).batch(BATCH_SIZE)\n\n# Print shapes of batches in train_ds\n# for batch in train_ds:\n#     print(batch[0].shape, batch[1].shape)\n    \n# Print shapes of batches in val_ds\n# for batch in val_ds:\n#     print(batch[0].shape, batch[1].shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:29:16.581381Z","iopub.execute_input":"2023-03-28T10:29:16.581783Z","iopub.status.idle":"2023-03-28T10:29:35.130513Z","shell.execute_reply.started":"2023-03-28T10:29:16.581744Z","shell.execute_reply":"2023-03-28T10:29:35.12902Z"},"trusted":true},"execution_count":29,"outputs":[{"name":"stdout","text":"train_idxs: <class 'numpy.ndarray'>\nval_idxs: <class 'numpy.ndarray'>\nval_x (9447, 5796)\nval_y (9447,)\ntrain_x (85030, 5796)\ntrain_y (85030,)\n","output_type":"stream"}]},{"cell_type":"code","source":"from kerastuner.tuners import RandomSearch, BayesianOptimization\nfrom tensorflow.keras.callbacks import EarlyStopping\n\ndef build_model(hp):\n    model = tf.keras.Sequential()\n    model.add(tf.keras.layers.Dense(hp.Int('units', min_value=256, max_value=1024, step=128), input_shape=(FLAT_FRAME_SHAPE,)))\n    model.add(tf.keras.layers.BatchNormalization())\n    model.add(tf.keras.layers.Activation(hp.Choice('activation', values=['relu', 'elu', 'selu', 'sigmoid'])))\n    model.add(tf.keras.layers.Dropout(hp.Float('dropout', min_value=0.2, max_value=0.5, step=0.1)))\n    \n    # Add more dense layers\n    for i in range(hp.Int('num_layers', 1, 3)):\n        model.add(tf.keras.layers.Dense(hp.Int(f'units_{i}', min_value=256, max_value=1024, step=128)))\n        model.add(tf.keras.layers.BatchNormalization())\n        model.add(tf.keras.layers.Activation(hp.Choice(f'activation_{i}', values=['relu', 'elu', 'selu', 'sigmoid'])))\n        model.add(tf.keras.layers.Dropout(hp.Float(f'dropout_{i}', min_value=0.2, max_value=0.5, step=0.1)))\n        N_LABELS = 250\n    model.add(tf.keras.layers.Dense(N_LABELS, activation='softmax'))\n\n    # Use a different optimizer\n    optimizer = hp.Choice('optimizer', values=['adam', 'rmsprop', 'sgd'])\n\n    model.compile(loss='sparse_categorical_crossentropy',\n                  optimizer=optimizer,\n                  metrics=['accuracy'])\n\n    return model\n\ntuner = BayesianOptimization(\n    build_model,\n    objective='val_accuracy',\n    max_trials=20,\n    executions_per_trial=1,\n    directory='/',\n    project_name='asl'\n)\n\n# Add early stopping\nes = EarlyStopping(monitor='val_loss', mode='min', patience=5)\n\ntuner.search(train_x, train_y, validation_data=(val_x, val_y), batch_size=32, epochs=50, callbacks=[es])\n\nbest_model = tuner.get_best_models(num_models=1)[0]\nbest_hyperparameters = tuner.get_best_hyperparameters(num_trials=1)[0]\n\nbest_model.summary()\nprint(best_hyperparameters)\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T11:00:17.13124Z","iopub.execute_input":"2023-03-28T11:00:17.131825Z"},"trusted":true},"execution_count":null,"outputs":[{"name":"stdout","text":"Trial 4 Complete [02h 07m 18s]\nval_accuracy: 0.6937652230262756\n\nBest val_accuracy So Far: 0.756007194519043\nTotal elapsed time: 06h 57m 42s\n\nSearch: Running Trial #5\n\nValue             |Best Value So Far |Hyperparameter\n1024              |1024              |units\nrelu              |relu              |activation\n0.5               |0.5               |dropout\n3                 |1                 |num_layers\n1024              |1024              |units_0\nrelu              |relu              |activation_0\n0.2               |0.2               |dropout_0\nadam              |adam              |optimizer\n1024              |640               |units_1\nsigmoid           |sigmoid           |activation_1\n0.5               |0.5               |dropout_1\n\nEpoch 1/50\n2953/2953 [==============================] - 321s 107ms/step - loss: 5.0576 - accuracy: 0.0271 - val_loss: 5.0335 - val_accuracy: 0.0267\nEpoch 2/50\n2953/2953 [==============================] - 301s 102ms/step - loss: 4.2238 - accuracy: 0.0850 - val_loss: 3.8492 - val_accuracy: 0.1452\nEpoch 3/50\n2953/2953 [==============================] - 302s 102ms/step - loss: 3.8423 - accuracy: 0.1327 - val_loss: 3.7636 - val_accuracy: 0.1319\nEpoch 4/50\n2953/2953 [==============================] - 305s 103ms/step - loss: 3.6267 - accuracy: 0.1648 - val_loss: 3.4239 - val_accuracy: 0.2057\nEpoch 5/50\n2953/2953 [==============================] - 307s 104ms/step - loss: 3.4495 - accuracy: 0.1950 - val_loss: 3.2004 - val_accuracy: 0.2552\nEpoch 6/50\n2953/2953 [==============================] - 300s 102ms/step - loss: 3.3329 - accuracy: 0.2165 - val_loss: 3.2525 - val_accuracy: 0.2305\nEpoch 7/50\n2953/2953 [==============================] - 298s 101ms/step - loss: 3.2432 - accuracy: 0.2331 - val_loss: 3.0654 - val_accuracy: 0.2650\nEpoch 8/50\n2953/2953 [==============================] - 295s 100ms/step - loss: 3.1688 - accuracy: 0.2480 - val_loss: 3.1431 - val_accuracy: 0.2528\nEpoch 9/50\n2953/2953 [==============================] - 304s 103ms/step - loss: 3.0998 - accuracy: 0.2611 - val_loss: 2.9297 - val_accuracy: 0.2954\nEpoch 10/50\n2953/2953 [==============================] - 306s 104ms/step - loss: 3.0427 - accuracy: 0.2702 - val_loss: 2.7958 - val_accuracy: 0.3288\nEpoch 11/50\n2953/2953 [==============================] - 305s 103ms/step - loss: 2.9870 - accuracy: 0.2827 - val_loss: 2.6888 - val_accuracy: 0.3563\nEpoch 12/50\n2953/2953 [==============================] - 310s 105ms/step - loss: 2.9363 - accuracy: 0.2936 - val_loss: 2.9366 - val_accuracy: 0.2978\nEpoch 13/50\n2953/2953 [==============================] - 319s 108ms/step - loss: 2.8940 - accuracy: 0.3008 - val_loss: 2.8379 - val_accuracy: 0.3078\nEpoch 14/50\n2953/2953 [==============================] - 310s 105ms/step - loss: 2.8377 - accuracy: 0.3115 - val_loss: 2.6008 - val_accuracy: 0.3594\nEpoch 15/50\n2953/2953 [==============================] - 313s 106ms/step - loss: 2.7950 - accuracy: 0.3202 - val_loss: 2.5621 - val_accuracy: 0.3692\nEpoch 16/50\n2953/2953 [==============================] - 307s 104ms/step - loss: 2.7607 - accuracy: 0.3291 - val_loss: 2.3717 - val_accuracy: 0.4307\nEpoch 17/50\n2953/2953 [==============================] - 303s 102ms/step - loss: 2.7211 - accuracy: 0.3350 - val_loss: 2.5652 - val_accuracy: 0.3670\nEpoch 18/50\n2953/2953 [==============================] - 297s 101ms/step - loss: 2.6848 - accuracy: 0.3441 - val_loss: 2.2704 - val_accuracy: 0.4374\nEpoch 19/50\n2953/2953 [==============================] - 297s 101ms/step - loss: 2.6527 - accuracy: 0.3493 - val_loss: 2.2489 - val_accuracy: 0.4570\nEpoch 20/50\n2953/2953 [==============================] - 301s 102ms/step - loss: 2.6231 - accuracy: 0.3574 - val_loss: 2.3556 - val_accuracy: 0.4230\nEpoch 21/50\n2953/2953 [==============================] - 308s 104ms/step - loss: 2.5979 - accuracy: 0.3615 - val_loss: 2.2183 - val_accuracy: 0.4556\nEpoch 22/50\n2953/2953 [==============================] - 312s 106ms/step - loss: 2.5676 - accuracy: 0.3691 - val_loss: 2.1720 - val_accuracy: 0.4710\nEpoch 23/50\n2953/2953 [==============================] - 298s 101ms/step - loss: 2.5483 - accuracy: 0.3721 - val_loss: 2.1109 - val_accuracy: 0.4822\nEpoch 24/50\n2953/2953 [==============================] - 298s 101ms/step - loss: 2.5278 - accuracy: 0.3779 - val_loss: 2.2445 - val_accuracy: 0.4441\nEpoch 25/50\n2953/2953 [==============================] - 298s 101ms/step - loss: 2.4954 - accuracy: 0.3862 - val_loss: 2.0438 - val_accuracy: 0.5082\nEpoch 26/50\n2953/2953 [==============================] - 306s 103ms/step - loss: 2.4677 - accuracy: 0.3899 - val_loss: 2.0541 - val_accuracy: 0.4930\nEpoch 27/50\n2953/2953 [==============================] - 295s 100ms/step - loss: 2.4546 - accuracy: 0.3928 - val_loss: 2.0281 - val_accuracy: 0.5002\nEpoch 28/50\n2953/2953 [==============================] - 303s 103ms/step - loss: 2.4355 - accuracy: 0.3990 - val_loss: 1.9969 - val_accuracy: 0.5112\nEpoch 29/50\n2953/2953 [==============================] - 295s 100ms/step - loss: 2.4164 - accuracy: 0.4009 - val_loss: 2.0543 - val_accuracy: 0.4965\nEpoch 30/50\n2953/2953 [==============================] - 298s 101ms/step - loss: 2.3899 - accuracy: 0.4069 - val_loss: 2.1191 - val_accuracy: 0.4695\nEpoch 31/50\n2953/2953 [==============================] - 294s 100ms/step - loss: 2.3721 - accuracy: 0.4101 - val_loss: 2.0732 - val_accuracy: 0.4872\nEpoch 32/50\n2953/2953 [==============================] - 301s 102ms/step - loss: 2.3470 - accuracy: 0.4156 - val_loss: 1.9109 - val_accuracy: 0.5210\nEpoch 33/50\n2953/2953 [==============================] - 304s 103ms/step - loss: 2.3353 - accuracy: 0.4176 - val_loss: 1.9199 - val_accuracy: 0.5284\nEpoch 34/50\n2953/2953 [==============================] - 301s 102ms/step - loss: 2.3228 - accuracy: 0.4194 - val_loss: 1.9546 - val_accuracy: 0.5131\nEpoch 35/50\n2953/2953 [==============================] - 300s 102ms/step - loss: 2.3078 - accuracy: 0.4239 - val_loss: 1.8745 - val_accuracy: 0.5446\nEpoch 36/50\n2953/2953 [==============================] - 294s 99ms/step - loss: 2.2848 - accuracy: 0.4287 - val_loss: 1.8379 - val_accuracy: 0.5448\nEpoch 37/50\n2953/2953 [==============================] - 298s 101ms/step - loss: 2.2595 - accuracy: 0.4342 - val_loss: 1.7884 - val_accuracy: 0.5622\nEpoch 38/50\n2953/2953 [==============================] - 298s 101ms/step - loss: 2.2588 - accuracy: 0.4357 - val_loss: 1.8693 - val_accuracy: 0.5243\nEpoch 39/50\n2953/2953 [==============================] - 298s 101ms/step - loss: 2.2446 - accuracy: 0.4363 - val_loss: 1.8127 - val_accuracy: 0.5428\nEpoch 40/50\n2953/2953 [==============================] - 302s 102ms/step - loss: 2.2250 - accuracy: 0.4416 - val_loss: 1.7240 - val_accuracy: 0.5697\nEpoch 41/50\n2953/2953 [==============================] - 304s 103ms/step - loss: 2.2150 - accuracy: 0.4443 - val_loss: 1.7536 - val_accuracy: 0.5663\nEpoch 42/50\n2885/2953 [============================>.] - ETA: 6s - loss: 2.2068 - accuracy: 0.4457","output_type":"stream"}]},{"cell_type":"code","source":"def load_training_data():\n    train_x = np.load(\"/kaggle/working/feature_data.npy\").astype(np.float32)\n    train_y = np.load(\"/kaggle/working/feature_labels.npy\").astype(np.uint8)\n    return train_x, train_y\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T21:49:04.884623Z","iopub.execute_input":"2023-03-28T21:49:04.885939Z","iopub.status.idle":"2023-03-28T21:49:04.892847Z","shell.execute_reply.started":"2023-03-28T21:49:04.885876Z","shell.execute_reply":"2023-03-28T21:49:04.891311Z"},"trusted":true},"execution_count":12,"outputs":[]},{"cell_type":"code","source":"# import tensorflow as tf\n# from tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\n# from tensorflow.keras.optimizers import Adam\n# from sklearn.model_selection import StratifiedKFold\n\n\n# # Define hyperparameters\n# LR_PATIENCE = 5\n# LR_REDUCTION_FACTOR = 0.1\n# STARTING_LAYER_SIZE = 1426\n# DROPOUTS_len = 2\n# dropout = [0.3, 0.3]\n# hp_activation = 'relu'\n# hp_learning_rate = 0.000333\n# hp_optimizer = 'adam'\n# EPOCHS = 50\n# BATCH_SIZE = 128\n# NUM_FOLDS = 5\n\n# # Define callbacks\n# cb_list = [\n#     EarlyStopping(patience=10, restore_best_weights=True, verbose=1),\n#     ReduceLROnPlateau(patience=LR_PATIENCE, factor=LR_REDUCTION_FACTOR, verbose=1)\n# ]\n\n# # Define optimizer\n# if hp_optimizer == 'lion':\n#     optimizer = Lion(hp_learning_rate)\n# elif hp_optimizer == 'adam':\n#     optimizer = Adam(hp_learning_rate)\n\n# # Define model architecture\n# def get_model(n_labels=250, init_fc=STARTING_LAYER_SIZE, flat_frame_len=FLAT_FRAME_SHAPE):\n#     _inputs = tf.keras.layers.Input(shape=(flat_frame_len,))\n#     x = _inputs\n\n#     for i in range(DROPOUTS_len):\n#         x = tf.keras.layers.Dense(init_fc // (2**i), activation=hp_activation)(x)\n#         x = tf.keras.layers.Dropout(dropout[i])(x)\n\n#     _outputs = tf.keras.layers.Dense(n_labels, activation=\"softmax\")(x)\n#     model = tf.keras.models.Model(inputs=_inputs, outputs=_outputs)\n#     return model\n\n# # Train model\n# train_x, train_y = load_training_data()  # Load training data\n# skf = StratifiedKFold(n_splits=NUM_FOLDS, shuffle=True, random_state=42)\n\n# for fold, (train_index, test_index) in enumerate(skf.split(train_x, train_y)):\n#     print(f\"Fold {fold}:\")\n\n#     val_x_fold, val_y_fold = train_x[test_index], train_y[test_index]\n#     train_x_fold, train_y_fold = train_x[train_index], train_y[train_index]\n\n#     model = get_model()\n#     model.compile(optimizer, \"sparse_categorical_crossentropy\", metrics=\"acc\")\n\n#     history = model.fit(train_x_fold, train_y_fold, validation_data=(val_x_fold, val_y_fold), epochs=EPOCHS,\n#                         callbacks=cb_list, batch_size=BATCH_SIZE)\n\n#     model.save(f\"./models/asl_model_{fold}\")\n\n#     # Evaluate model\n#     loss, acc = model.evaluate(val_x_fold, val_y_fold)\n#     print(f\"Validation loss: {loss}, Validation accuracy: {acc}\")\n\n#     # Make predictions\n#     predictions = model.predict(val_x_fold)\n#     for i in range(10):\n#         pred_label = decoder(tf.argmax(predictions[i], axis=-1))\n#         true_label = decoder(val_y_fold[i])\n#         print(f\"Prediction: {pred_label:<20} – Ground Truth: {true_label}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:29:35.14286Z","iopub.execute_input":"2023-03-28T10:29:35.143325Z","iopub.status.idle":"2023-03-28T10:29:35.16612Z","shell.execute_reply.started":"2023-03-28T10:29:35.143273Z","shell.execute_reply":"2023-03-28T10:29:35.163798Z"},"trusted":true},"execution_count":31,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\nfrom tensorflow.keras.optimizers import Adam\nfrom sklearn.model_selection import StratifiedKFold\n\n\n# Define hyperparameters\nLR_PATIENCE = 5\nLR_REDUCTION_FACTOR = 0.1\nSTARTING_LAYER_SIZE = 2048\nDROPOUTS_len = 2\ndropout = [0.3, 0.3]\nhp_activation = 'relu'\nhp_learning_rate = 0.001\nhp_optimizer = 'adam'\nEPOCHS = 50\nBATCH_SIZE = 128\nNUM_FOLDS = 5\n\n# Define callbacks\ncb_list = [\n    EarlyStopping(patience=10, restore_best_weights=True, verbose=1),\n    ReduceLROnPlateau(patience=LR_PATIENCE, factor=LR_REDUCTION_FACTOR, verbose=1)\n]\n\n# Define optimizer\nif hp_optimizer == 'lion':\n    optimizer = Lion(hp_learning_rate)\nelif hp_optimizer == 'adam':\n    optimizer = Adam(hp_learning_rate)\n\n# Define model architecture\ndef get_model(n_labels=250, init_fc=STARTING_LAYER_SIZE, flat_frame_len=FLAT_FRAME_SHAPE):\n    _inputs = tf.keras.layers.Input(shape=(flat_frame_len,))\n    x = _inputs\n    x = tf.keras.layers.BatchNormalization()(x)\n\n    for i in range(DROPOUTS_len):\n        x = tf.keras.layers.Dense(init_fc // (2**i), activation=hp_activation)(x)\n        x = tf.keras.layers.BatchNormalization()(x)\n        x = tf.keras.layers.Dropout(dropout[i])(x)\n\n    _outputs = tf.keras.layers.Dense(n_labels, activation=\"softmax\")(x)\n    model = tf.keras.models.Model(inputs=_inputs, outputs=_outputs)\n    return model\n\n# Train model\ntrain_x, train_y = load_training_data()  # Load training data\nskf = StratifiedKFold(n_splits=NUM_FOLDS, shuffle=True, random_state=42)\n\nfor fold, (train_index, test_index) in enumerate(skf.split(train_x, train_y)):\n    print(f\"Fold {fold}:\")\n\n    val_x_fold, val_y_fold = train_x[test_index], train_y[test_index]\n    train_x_fold, train_y_fold = train_x[train_index], train_y[train_index]\n\n    model = get_model()\n    model.compile(optimizer, \"sparse_categorical_crossentropy\", metrics=[\"acc\"])\n\n    history = model.fit(train_x_fold, train_y_fold, validation_data=(val_x_fold, val_y_fold), epochs=EPOCHS,\n                        callbacks=cb_list, batch_size=BATCH_SIZE)\n\n    model.save(f\"./models/asl_model_{fold}\")\n\n    # Evaluate model\n    loss, acc = model.evaluate(val_x_fold, val_y_fold)\n    print(f\"Validation loss: {loss}, Validation accuracy: {acc}\")\n\n    # Make predictions\n    predictions = model.predict(val_x_fold)\n    for i in range(10):\n        pred_label = decoder(tf.argmax(predictions[i], axis=-1))\n        true_label = decoder(val_y_fold[i])\n        print(f\"Prediction: {pred_label:<20} – Ground Truth: {true_label}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T21:49:08.03539Z","iopub.execute_input":"2023-03-28T21:49:08.035827Z","iopub.status.idle":"2023-03-28T21:49:08.089533Z","shell.execute_reply.started":"2023-03-28T21:49:08.035789Z","shell.execute_reply":"2023-03-28T21:49:08.087846Z"},"trusted":true},"execution_count":13,"outputs":[{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mNameError\u001b[0m                                 Traceback (most recent call last)","\u001b[0;32m/tmp/ipykernel_27/2119954919.py\u001b[0m in \u001b[0;36m<module>\u001b[0;34m\u001b[0m\n\u001b[1;32m     31\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     32\u001b[0m \u001b[0;31m# Define model architecture\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 33\u001b[0;31m \u001b[0;32mdef\u001b[0m \u001b[0mget_model\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mn_labels\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;36m250\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0minit_fc\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mSTARTING_LAYER_SIZE\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mflat_frame_len\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0mFLAT_FRAME_SHAPE\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m:\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m     34\u001b[0m     \u001b[0m_inputs\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mtf\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mkeras\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mlayers\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mInput\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mshape\u001b[0m\u001b[0;34m=\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mflat_frame_len\u001b[0m\u001b[0;34m,\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     35\u001b[0m     \u001b[0mx\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0m_inputs\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;31mNameError\u001b[0m: name 'FLAT_FRAME_SHAPE' is not defined"],"ename":"NameError","evalue":"name 'FLAT_FRAME_SHAPE' is not defined","output_type":"error"}]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install lion-tf\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:50:34.097414Z","iopub.status.idle":"2023-03-28T10:50:34.098181Z","shell.execute_reply.started":"2023-03-28T10:50:34.097854Z","shell.execute_reply":"2023-03-28T10:50:34.097888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from kerastuner.tuners import BayesianOptimization\n# from lion_tf import Lion\n\n# def fc_block(inputs, output_channels, dropout=0.2, activation=\"gelu\"):\n#     x = tf.keras.layers.Dense(output_channels)(inputs)\n#     x = tf.keras.layers.BatchNormalization()(x)\n#     x = tf.keras.layers.Activation(activation)(x)\n#     x = tf.keras.layers.Dropout(dropout)(x)\n#     return x\n\n# def build_model(hp):\n    \n#     STARTING_LAYER_SIZE = hp.Int(name=\"STARTING_LAYER_SIZE\", min_value=768, max_value=1536, step=128)\n#     dropout = hp.Float(name=\"dropout\", min_value=0.3, max_value=0.5, step=0.01)\n#     DROPOUTS_len = hp.Int(name=\"dropouts_len\", min_value=2, max_value=5, step=1)\n#     hp_activation = hp.Choice('hp_activation', values=[\"gelu\", \"relu\"])\n#     hp_learning_rate = hp.Choice('hp_learning_rate', values=[1e-4, 1e-5, 1e-6])\n    \n#     hp_optimizer = hp.Choice('optimizer', values=['adam', 'lion'])\n\n#     if hp_optimizer == 'lion':\n#         optimizer = Lion(hp_learning_rate)\n#     elif hp_optimizer == 'adam':\n#         optimizer = tf.keras.optimizers.Adam(hp_learning_rate)\n        \n#     def get_model(n_labels=250, init_fc=STARTING_LAYER_SIZE, flat_frame_len=FLAT_FRAME_SHAPE):\n#         _inputs = tf.keras.layers.Input(shape=(flat_frame_len,))\n#         x = _inputs\n        \n#         for i in range(DROPOUTS_len):\n#             x = fc_block(x, output_channels=init_fc//(2**i), dropout=dropout)\n\n#         _outputs = tf.keras.layers.Dense(n_labels, activation=\"softmax\")(x)\n#         model = tf.keras.models.Model(inputs=_inputs, outputs=_outputs)\n#         return model\n\n#     model = get_model()\n#     model.compile(optimizer, \"sparse_categorical_crossentropy\", metrics=[\"accuracy\"])\n#     model.summary()\n#     return model\n\n# tuner = BayesianOptimization(\n#     build_model,\n#     objective='val_accuracy',\n#     max_trials=20,\n#     executions_per_trial=1,\n#     directory='/',\n#     project_name='asl'\n# )\n","metadata":{"execution":{"iopub.status.busy":"2023-03-28T10:50:34.101329Z","iopub.status.idle":"2023-03-28T10:50:34.10199Z","shell.execute_reply.started":"2023-03-28T10:50:34.101697Z","shell.execute_reply":"2023-03-28T10:50:34.101731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}