{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# import","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nfrom matplotlib.animation import FuncAnimation\nfrom IPython.display import HTML\nimport os\nimport json\nfrom glob import glob\nimport math\nimport multiprocessing as mp\nfrom tqdm import tqdm\nimport random\n\nimport tensorflow as tf; print(f\"\\t\\t– TENSORFLOW VERSION: {tf.__version__}\");\nimport tensorflow_io as tfio; print(f\"\\t\\t– TENSORFLOW-IO VERSION: {tfio.__version__}\");\n\n# set the display options to show all columns\npd.set_option('display.max_columns', 100)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:46:54.842186Z","iopub.execute_input":"2023-04-19T05:46:54.842611Z","iopub.status.idle":"2023-04-19T05:47:06.602421Z","shell.execute_reply.started":"2023-04-19T05:46:54.842575Z","shell.execute_reply":"2023-04-19T05:47:06.600733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper functions","metadata":{}},{"cell_type":"code","source":"def read_json_file(file_path):\n    \"\"\"Read a JSON file and parse it into a Python object.\n\n    Args:\n        file_path (str): The path to the JSON file to read.\n\n    Returns:\n        dict: A dictionary object representing the JSON data.\n        \n    Raises:\n        FileNotFoundError: If the specified file path does not exist.\n        ValueError: If the specified file path does not contain valid JSON data.\n    \"\"\"\n    try:\n        # Open the file and load the JSON data into a Python object\n        with open(file_path, 'r') as file:\n            json_data = json.load(file)\n        return json_data\n    except FileNotFoundError:\n        # Raise an error if the file path does not exist\n        raise FileNotFoundError(f\"File not found: {file_path}\")\n    except ValueError:\n        # Raise an error if the file does not contain valid JSON data\n        raise ValueError(f\"Invalid JSON data in file: {file_path}\")\n        \ndef get_sign_df(pq_path, invert_y=True):\n    sign_df = pd.read_parquet(pq_path)\n    \n    # y value is inverted (Thanks @danielpeshkov)\n    if invert_y: sign_df[\"y\"] *= -1 \n        \n    return sign_df\n\ndef get_hand_points(hand):\n    \"\"\"Return x, y lists of normalized spatial coordinates for each finger in the hand dataframe.\"\"\"\n    def __get_hand_ax(_axis):\n        return [np.nan_to_num(_x) for _x in \n            [hand.iloc[i][_axis] for i in range(5)]+\\\n            [[hand.iloc[i][_axis] for i in range(j, j+4)] for j in range(5, 21, 4)]+\\\n            [hand.iloc[i][_axis] for i in special_pts]]\n    special_pts = [0, 5, 9, 13, 17, 0]\n    return [__get_hand_ax(_ax) for _ax in ['x','y','z']]\n\ndef get_pose_points(pose):\n    \"\"\"\n    Extracts x and y coordinates from the provided dataframe for pose landmarks.\n\n    Args:\n        pose (pandas.DataFrame): Dataframe containing pose landmarks with columns ['x', 'y', 'z', 'visibility', 'presence'].\n\n    Returns:\n        tuple: Two lists of x and y coordinates, respectively.\n\n    \"\"\"\n    def __get_pose_ax(_axis):\n        return [np.nan_to_num(_x) for _x in [\n            [pose.iloc[i][_axis] for i in [8, 6, 5, 4, 0, 1, 2, 3, 7]], \n            [pose.iloc[i][_axis] for i in [10, 9]], \n            [pose.iloc[i][_axis] for i in [22, 16, 20, 18, 16, 14, 12, 11, 13, 15, 17, 19, 15, 21]], \n            [pose.iloc[i][_axis] for i in [12, 24, 26, 28, 30, 32, 28]], \n            [pose.iloc[i][_axis] for i in [11, 23, 25, 27, 29, 31, 27]], \n            [pose.iloc[i][_axis] for i in [24, 23]]\n        ]]\n    return [__get_pose_ax(_ax) for _ax in ['x','y','z']]\n\ndef animation_frame(f, event_df, ax, ax_pad=0.2, style=\"full\", \n                    face_color=\"spring\", pose_color=\"autumn\", lh_color=\"winter\", rh_color=\"summer\"):\n    \"\"\"\n    Function called by FuncAnimation to animate the plot with the provided frame.\n\n    Args:\n        f (int): The current frame number.\n\n    Returns:\n        None.\n    \"\"\"\n    \n    face_color = plt.cm.get_cmap(face_color)\n    pose_color = plt.cm.get_cmap(pose_color)\n    rh_color = plt.cm.get_cmap(rh_color)\n    lh_color = plt.cm.get_cmap(lh_color)\n    \n    sign_df = event_df.copy()\n    \n    # Clear axis and fix the axis\n    ax.clear()\n    if style==\"full\":\n        xmin = sign_df['x'].min() - ax_pad\n        xmax = sign_df['x'].max() + ax_pad\n        ymin = sign_df['y'].min() - ax_pad\n        ymax = sign_df['y'].max() + ax_pad\n    elif style==\"hands\":\n        xmin = sign_df[sign_df.type.isin([\"left_hand\", \"right_hand\"])]['x'].min() - ax_pad\n        xmax = sign_df[sign_df.type.isin([\"left_hand\", \"right_hand\"])]['x'].max() + ax_pad\n        ymin = sign_df[sign_df.type.isin([\"left_hand\", \"right_hand\"])]['y'].min() - ax_pad\n        ymax = sign_df[sign_df.type.isin([\"left_hand\", \"right_hand\"])]['y'].max() + ax_pad\n    else:\n        xmin = sign_df[sign_df.type==style]['x'].min() - ax_pad\n        xmax = sign_df[sign_df.type==style]['x'].max() + ax_pad\n        ymin = sign_df[sign_df.type==style]['y'].min() - ax_pad\n        ymax = sign_df[sign_df.type==style]['y'].max() + ax_pad\n    \n    ax.set_xlim(xmin, xmax)\n    ax.set_ylim(ymin, ymax)\n    ax.axis(False) # Remove the axis lines\n    \n    # Normalize depth\n    zmin, zmax = sign_df['z'].min(), sign_df['z'].max()\n    sign_df['z'] = (sign_df['z']-zmin)/(zmax-zmin)\n    \n    # Get data for current frame\n    frame = sign_df[sign_df.frame==f]\n    \n    # Left Hand\n    if style.lower() in [\"left_hand\", \"hands\", \"full\"]:\n        left = frame[frame.type=='left_hand']\n        lx, ly, lz = get_hand_points(left)\n        for i in range(len(lx)):\n            if type(lx[i])!=np.float64:\n                lh_clr = [lh_color(((np.abs(_x)+np.abs(_y))/2)) for _x, _y in zip(lx[i], ly[i])]\n                lh_clr = tuple(sum(_x)/len(_x) for _x in zip(*lh_clr))\n            else: \n                lh_clr = lh_color(((np.abs(lx[i])+np.abs(ly[i]))/2))\n            ax.plot(lx[i], ly[i], color=lh_clr, alpha=lz[i].mean())\n    \n    # Right Hand\n    if style.lower() in [\"right_hand\", \"hands\", \"full\"]:\n        right = frame[frame.type=='right_hand']\n        rx, ry, rz = get_hand_points(right)\n        for i in range(len(rx)):\n            if type(rx[i])!=np.float64:\n                rh_clr = [rh_color((np.abs(_x)+np.abs(_y))/2) for _x, _y in zip(rx[i], ry[i])] \n                rh_clr = tuple(sum(_x)/len(_x) for _x in zip(*rh_clr))\n            else:\n                rh_clr = rh_color(((np.abs(rx[i])+np.abs(ry[i]))/2))\n            ax.plot(rx[i], ry[i], color=rh_clr, alpha=rz[i].mean())\n    \n    # Pose\n    if style.lower() in [\"pose\", \"full\"]:\n        pose = frame[frame.type=='pose']\n        px, py, pz = get_pose_points(pose)\n        for i in range(len(px)):\n            if type(px[i])!=np.float64:\n                pose_clr = [pose_color(((np.abs(_x)+np.abs(_y))/2)) for _x, _y in zip(px[i], py[i])]\n                pose_clr = tuple(sum(_x)/len(_x) for _x in zip(*pose_clr))\n            else: \n                pose_clr = pose_color(((np.abs(px[i])+np.abs(py[i]))/2))\n            ax.plot(px[i], py[i], color=pose_clr, alpha=pz[i].mean())\n        \n    if style.lower() in [\"face\", \"full\"]:\n        face = frame[frame.type=='face'][['x', 'y', 'z']].values\n        fx, fy, fz = face[:,0], face[:,1], face[:,2]\n        for i in range(len(fx)):\n            ax.plot(fx[i], fy[i], '.', color=pose_color(fz[i]), alpha=fz[i])\n    \n    # Use this so we don't get an extra return\n    plt.close()    \n    \ndef plot_event(event_df, style=\"full\"):\n    # Create figure and animation\n    fig, ax = plt.subplots()\n    l, = ax.plot([], [])\n    animation = FuncAnimation(fig, func=lambda x: animation_frame(x, event_df, ax, style=style), \n                              frames=event_df[\"frame\"].unique())\n    \n    # Display animation as HTML5 video\n    return HTML(animation.to_html5_video())\n","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:06.605556Z","iopub.execute_input":"2023-04-19T05:47:06.606897Z","iopub.status.idle":"2023-04-19T05:47:06.654122Z","shell.execute_reply.started":"2023-04-19T05:47:06.606839Z","shell.execute_reply":"2023-04-19T05:47:06.652692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Read Input\n\n- base directory -> `/kaggle/input/asl-signs`\n- participant_id -> `.../16069`\n- sequence_id -> `.../100015657.parquet`","metadata":{}},{"cell_type":"code","source":"# Define the path to the root data directory\nDATA_DIR         = \"/kaggle/input/asl-signs\"\n# EXTEND_TRAIN_DIR = \"/kaggle/input/gislr-extended-train-dataframe\" \n\nprint(\"\\n... BASIC DATA SETUP STARTING ...\\n\")\nprint(\"\\n\\n... LOAD TRAIN DATAFRAME FROM CSV FILE ...\\n\")\n\ntrain_df = pd.read_csv(os.path.join(DATA_DIR, 'train.csv'))\ntrain_df[\"participant_id\"] = train_df[\"participant_id\"].astype('string')\ndisplay(train_df)\n\nprint(\"\\n\\n... LOAD SIGN TO PREDICTION INDEX MAP FROM JSON FILE ...\\n\")\n\ns2p_map = {k.lower():v for k,v in read_json_file(os.path.join(DATA_DIR, \"sign_to_prediction_index_map.json\")).items()}\np2s_map = {v:k for k,v in read_json_file(os.path.join(DATA_DIR, \"sign_to_prediction_index_map.json\")).items()}\nencoder = lambda x: s2p_map.get(x.lower())\ndecoder = lambda x: p2s_map.get(x)\nprint(s2p_map)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-19T05:47:06.655802Z","iopub.execute_input":"2023-04-19T05:47:06.656665Z","iopub.status.idle":"2023-04-19T05:47:07.034712Z","shell.execute_reply.started":"2023-04-19T05:47:06.656620Z","shell.execute_reply":"2023-04-19T05:47:07.033504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sample parquet","metadata":{}},{"cell_type":"code","source":"DEMO_ROW = 283\nprint(f\"\\n\\n... DEMO SIGN/EVENT DATAFRAME FOR ROW {DEMO_ROW} - SIGN={train_df.iloc[DEMO_ROW]['sign']} ...\\n\")\ndemo_sign_df = get_sign_df(os.path.join(DATA_DIR, train_df.iloc[DEMO_ROW][\"path\"]))\ndisplay(demo_sign_df)\n\nplot_event(demo_sign_df)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:07.037671Z","iopub.execute_input":"2023-04-19T05:47:07.038053Z","iopub.status.idle":"2023-04-19T05:47:20.546959Z","shell.execute_reply.started":"2023-04-19T05:47:07.038019Z","shell.execute_reply":"2023-04-19T05:47:20.545528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sample frame\n\n- face ->         468 \n- pose ->          33 \n- left_hand ->     21 \n- right_hand ->    21 \n\n\n\n- The frame without left_hand and right_hand may be remove??\n","metadata":{}},{"cell_type":"code","source":"demo_sign_df.query('frame == 24')","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:20.548788Z","iopub.execute_input":"2023-04-19T05:47:20.549840Z","iopub.status.idle":"2023-04-19T05:47:20.577359Z","shell.execute_reply.started":"2023-04-19T05:47:20.549786Z","shell.execute_reply":"2023-04-19T05:47:20.576185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## interest to explore\n\nThere some missing values.\n- at frame 35, only 501 position values\n- total 543 points, but maxium only 522 position values","metadata":{}},{"cell_type":"code","source":"print('... min and max frames ...\\n')\n\nframe_min, frame_max = demo_sign_df['frame'].min(), demo_sign_df['frame'].max()\nprint('min: {}, max: {}'.format(frame_min, frame_max))\n\nprint('\\n\\n... all points each frame ...\\n')\n\nprint(demo_sign_df.groupby('frame').count())\n\nprint('\\n\\n... points each type ...\\n')\n\nprint(demo_sign_df.query('frame == 23')['type'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:20.579046Z","iopub.execute_input":"2023-04-19T05:47:20.579408Z","iopub.status.idle":"2023-04-19T05:47:20.607388Z","shell.execute_reply.started":"2023-04-19T05:47:20.579375Z","shell.execute_reply":"2023-04-19T05:47:20.606014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Chek frame 35\n\n- left_hand and right_hand doesn't have position values[x, y, z].","metadata":{}},{"cell_type":"code","source":"# demo_sign_df.query('frame == 35')[demo_sign_df.query('frame == 35')['x'].isna()]\nfilter = demo_sign_df.loc[demo_sign_df['frame'] == 35, 'x'].isna()\ndemo_sign_df.query('frame == 35')[filter]\n","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:20.609332Z","iopub.execute_input":"2023-04-19T05:47:20.610132Z","iopub.status.idle":"2023-04-19T05:47:20.638497Z","shell.execute_reply.started":"2023-04-19T05:47:20.610080Z","shell.execute_reply":"2023-04-19T05:47:20.637277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Chek frame 24\n\n- right_hand doesn't have position values[x, y, z].","metadata":{}},{"cell_type":"code","source":"filter = demo_sign_df.loc[demo_sign_df['frame'] == 24, 'x'].isna()\ndemo_sign_df.query('frame == 24')[filter]","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:20.639863Z","iopub.execute_input":"2023-04-19T05:47:20.640202Z","iopub.status.idle":"2023-04-19T05:47:20.663702Z","shell.execute_reply.started":"2023-04-19T05:47:20.640162Z","shell.execute_reply":"2023-04-19T05:47:20.662340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### check NaN\n\n- no data at `right_hand` \n- frame 35 has no data at `left_hand` ","metadata":{}},{"cell_type":"code","source":"ax = demo_sign_df.query('type == \"right_hand\"').groupby('frame')['x'].count().plot(kind='bar', \n                                                                              title='points of right_hand', \n                                                                              ylabel='count')\nax.set_ylim(bottom=0)\n# ax.yaxis.set_major_locator(plt.MaxNLocator(integer=True))\n# ax.yaxis.set_major_formatter('{:.2f}'.format)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:20.666346Z","iopub.execute_input":"2023-04-19T05:47:20.666717Z","iopub.status.idle":"2023-04-19T05:47:20.961281Z","shell.execute_reply.started":"2023-04-19T05:47:20.666684Z","shell.execute_reply":"2023-04-19T05:47:20.960093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ax = demo_sign_df.query('type == \"left_hand\"').groupby('frame')['x'].count().plot(kind='bar', \n                                                                             title='points of left_hand', \n                                                                             ylabel='count')\nax.yaxis.set_major_locator(plt.MaxNLocator(integer=True))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:20.967198Z","iopub.execute_input":"2023-04-19T05:47:20.967641Z","iopub.status.idle":"2023-04-19T05:47:21.248011Z","shell.execute_reply.started":"2023-04-19T05:47:20.967603Z","shell.execute_reply":"2023-04-19T05:47:21.246984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# pd.unique(demo_sign_df['type'])\n# demo_sign_df.query('frame == 24')\n# pd.unique(demo_sign_df['frame'])\n\n# tmp = demo_sign_df.query('frame == 24')\n# tmp[demo_sign_df['x'].isna()]","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:21.249391Z","iopub.execute_input":"2023-04-19T05:47:21.250191Z","iopub.status.idle":"2023-04-19T05:47:21.255817Z","shell.execute_reply.started":"2023-04-19T05:47:21.250150Z","shell.execute_reply":"2023-04-19T05:47:21.254233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data (train.csv)","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:21.258180Z","iopub.execute_input":"2023-04-19T05:47:21.258952Z","iopub.status.idle":"2023-04-19T05:47:21.273502Z","shell.execute_reply.started":"2023-04-19T05:47:21.258905Z","shell.execute_reply":"2023-04-19T05:47:21.272159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Path","metadata":{}},{"cell_type":"code","source":"print(\"\\n... With duplicated:\")\nprint(train_df.duplicated().sum())\n\ndisplay(train_df[\"path\"].to_frame().describe())","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:21.275063Z","iopub.execute_input":"2023-04-19T05:47:21.275940Z","iopub.status.idle":"2023-04-19T05:47:21.436776Z","shell.execute_reply.started":"2023-04-19T05:47:21.275898Z","shell.execute_reply":"2023-04-19T05:47:21.435513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## patricipants","metadata":{}},{"cell_type":"code","source":"print(\"... Number of Paticipants\")\nprint(len(train_df['participant_id'].unique()))\ntrain_df['participant_id'].value_counts().plot(kind='bar', \n                                               figsize=(8, 5), \n                                               title='Row counts by participant', \n                                               xlabel='participant ID', \n                                               ylabel='frequancy')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:21.438429Z","iopub.execute_input":"2023-04-19T05:47:21.438828Z","iopub.status.idle":"2023-04-19T05:47:21.771087Z","shell.execute_reply.started":"2023-04-19T05:47:21.438791Z","shell.execute_reply":"2023-04-19T05:47:21.769660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Sign (label)\n\n- dataset are balance","metadata":{}},{"cell_type":"code","source":"display(train_df[\"sign\"].to_frame().describe().T)\n\nsign_count = train_df['sign'].value_counts()\n\nprint(\"\\n\\n ... Rows count by sign\\n\")\n\n# sign_count.plot(kind='barh', figsize=(8, 50), title='Rows count by sign')\n# plt.show()\n\nfig = px.histogram(train_df, \n                   y=train_df[\"sign\"], \n                   color=\"sign\", \n                   orientation=\"h\", \n                   height=5000,\n                   labels={\"y\":\"<b>Sign (label)</b>\", \"count\":\"<b>Total Row Count</b>\"}, \n                   title=\"<b>Row Counts by Sign (label)</b>\",\n                   category_orders={\"sign\": sign_count.index}\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:21.773044Z","iopub.execute_input":"2023-04-19T05:47:21.773877Z","iopub.status.idle":"2023-04-19T05:47:24.810342Z","shell.execute_reply.started":"2023-04-19T05:47:21.773826Z","shell.execute_reply":"2023-04-19T05:47:24.809275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('... basic statistics of sign count map ...\\n')\n\nprint('\\t1. mean of sign count                --> {}'.format(sign_count.mean()))\nprint('\\t2. standard deviation of sign count  --> {}'.format(sign_count.std()))\nprint('\\t3. maximun of sign count:            --> {}'.format(sign_count.max()))\nprint('\\t4. minimun of sign count             --> {}'.format(sign_count.min()))\n\n# sign_count.plot(kind='box', title='count of sign distribution')\n# plt.show()\n\nprint('\\n\\n... box plot of sign count map ...\\n')\n\nfig = px.box(train_df, y=sign_count)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:24.811707Z","iopub.execute_input":"2023-04-19T05:47:24.812536Z","iopub.status.idle":"2023-04-19T05:47:24.911391Z","shell.execute_reply.started":"2023-04-19T05:47:24.812487Z","shell.execute_reply":"2023-04-19T05:47:24.910014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analaysis on parquet files","metadata":{}},{"cell_type":"code","source":"def get_meta(row):\n    cur_df = get_sign_df(os.path.join(DATA_DIR, row['path']))\n    row['start_frame'] = cur_df['frame'].min()\n    row['end_frame'] = cur_df['frame'].max()\n    row['total_frame'] = len(cur_df['frame'].unique())\n\n    type_counts = cur_df['type'].value_counts()\n    nan_counts = cur_df.groupby('type')['x'].apply(lambda x: x.isna().sum())\n\n    for _type in ['face', 'pose', 'left_hand', 'right_hand']:\n        row[f'{_type}_counts'] = type_counts[_type]\n        row[f'{_type}_nan_counts'] = nan_counts[_type]\n        \n    for _coord in ['x', 'y', 'z']:\n        row[f'{_coord}_min'] = cur_df[_coord].min()\n        row[f'{_coord}_max'] = cur_df[_coord].max()\n    \n    return row","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:24.913115Z","iopub.execute_input":"2023-04-19T05:47:24.913538Z","iopub.status.idle":"2023-04-19T05:47:24.922901Z","shell.execute_reply.started":"2023-04-19T05:47:24.913497Z","shell.execute_reply":"2023-04-19T05:47:24.921633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_df = train_df[0:math.floor(train_df.shape[0] * 0.0002)]\ntmp_df = tmp_df.apply(get_meta, axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:24.924128Z","iopub.execute_input":"2023-04-19T05:47:24.925140Z","iopub.status.idle":"2023-04-19T05:47:25.974012Z","shell.execute_reply.started":"2023-04-19T05:47:24.925100Z","shell.execute_reply":"2023-04-19T05:47:25.973065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get nan percentige\ndef get_nan_pct(df):\n    for _type in ['face', 'pose', 'left_hand', 'right_hand']:\n        df[f'{_type}_nan_pct'] = df[f'{_type}_nan_counts'] / df[f'{_type}_counts']\n    \n    return df\n    \ntmp_df = get_nan_pct(tmp_df)\ndisplay(tmp_df)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:25.975308Z","iopub.execute_input":"2023-04-19T05:47:25.975893Z","iopub.status.idle":"2023-04-19T05:47:26.013109Z","shell.execute_reply.started":"2023-04-19T05:47:25.975855Z","shell.execute_reply":"2023-04-19T05:47:26.011865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## NaN percentage of coordinate\n\n- face have some NaN\n- pose are never NaN\n","metadata":{}},{"cell_type":"code","source":"# matplotlib\n# tmp_df['face_nan_pct'].plot(kind='hist')\n# plt.show()\n# tmp_df['pose_nan_pct'].plot(kind='hist')\n# plt.show()\n# tmp_df['left_hand_nan_pct'].plot(kind='hist')\n# plt.show()\n# tmp_df['right_hand_nan_pct'].plot(kind='hist')\n# plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:26.014551Z","iopub.execute_input":"2023-04-19T05:47:26.014938Z","iopub.status.idle":"2023-04-19T05:47:26.020333Z","shell.execute_reply.started":"2023-04-19T05:47:26.014901Z","shell.execute_reply":"2023-04-19T05:47:26.018839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.histogram(tmp_df, \n                   [\"face_nan_pct\", \"left_hand_nan_pct\", \"pose_nan_pct\", \"right_hand_nan_pct\"], \n                   height=750, \n                   width=1000,\n                   nbins= 20,\n                   labels={'variable': '', 'value':\"<b>Percentage of Points That Are NaN</b>\"},\n#                    log_y=True,\n                   facet_col='variable',\n                   facet_col_wrap=2, \n                   facet_col_spacing=0.05)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:26.021944Z","iopub.execute_input":"2023-04-19T05:47:26.022701Z","iopub.status.idle":"2023-04-19T05:47:26.182082Z","shell.execute_reply.started":"2023-04-19T05:47:26.022661Z","shell.execute_reply":"2023-04-19T05:47:26.180535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# frame box plot","metadata":{}},{"cell_type":"code","source":"fig = px.box(tmp_df, \n             y=['start_frame', 'end_frame', 'total_frame'], \n             title='<b>plot start_frame, end_frame and total_frame</b>', \n             labels={'variable': 'kind of frame', 'value':\"<b>number of frames</b>\"})\n\n# Customize the box and whisker colors and width\nfig.update_traces(boxmean=True)\n\n# fig.update_xaxes(title_text='<b>Frame Measure</b>')\n# fig.update_yaxes(title_text='<b>Number of Frames</b>')\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:26.183927Z","iopub.execute_input":"2023-04-19T05:47:26.184307Z","iopub.status.idle":"2023-04-19T05:47:26.255163Z","shell.execute_reply.started":"2023-04-19T05:47:26.184271Z","shell.execute_reply":"2023-04-19T05:47:26.253767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:26.256983Z","iopub.execute_input":"2023-04-19T05:47:26.257824Z","iopub.status.idle":"2023-04-19T05:47:26.276187Z","shell.execute_reply.started":"2023-04-19T05:47:26.257770Z","shell.execute_reply":"2023-04-19T05:47:26.274911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"demo_sign_df","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:26.278037Z","iopub.execute_input":"2023-04-19T05:47:26.278549Z","iopub.status.idle":"2023-04-19T05:47:26.301654Z","shell.execute_reply.started":"2023-04-19T05:47:26.278498Z","shell.execute_reply":"2023-04-19T05:47:26.300496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tmp_df","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:26.302868Z","iopub.execute_input":"2023-04-19T05:47:26.303276Z","iopub.status.idle":"2023-04-19T05:47:26.341465Z","shell.execute_reply.started":"2023-04-19T05:47:26.303212Z","shell.execute_reply":"2023-04-19T05:47:26.339229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create features\n- sperate by lips, left_hand, right_hand, pose\n- mean and std of (x, y, z) of point that is group by different part(lips, left_hand, right_hand, pose).","metadata":{}},{"cell_type":"code","source":"label_map = json.load(open(\"/kaggle/input/asl-signs/sign_to_prediction_index_map.json\", \"r\"))\n\n# label_map = {k: int(v) for k, v in label_map.items()}\nprint(type(label_map.get('TV')))","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:26.343201Z","iopub.execute_input":"2023-04-19T05:47:26.343605Z","iopub.status.idle":"2023-04-19T05:47:26.350190Z","shell.execute_reply.started":"2023-04-19T05:47:26.343566Z","shell.execute_reply":"2023-04-19T05:47:26.349318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROWS_PER_FRAME = 468 + 33 + 21 + 21\n\n# https://www.kaggle.com/competitions/asl-signs/discussion/391812#2168354\nlipsUpperOuter =  [61, 185, 40, 39, 37, 0, 267, 269, 270, 409, 291]\nlipsLowerOuter = [146, 91, 181, 84, 17, 314, 405, 321, 375, 291]\nlipsUpperInner = [78, 191, 80, 81, 82, 13, 312, 311, 310, 415, 308]\nlipsLowerInner = [78, 95, 88, 178, 87, 14, 317, 402, 318, 324, 308]\nlips = lipsUpperOuter + lipsLowerOuter + lipsUpperInner + lipsLowerInner\n\nN_LIPS = 43\nN_HAND = 21\nN_POSE = 33\n# mean and std of (x, y, z) of point that is group by different part(lips, left_hand, right_hand, pose).\nFEATURE_SIZE = (N_LIPS * 3 + N_HAND * 3 * 2 + N_POSE * 3) * 2\n\ndef create_features(row):\n    feat_columns = ['x', 'y', 'z']\n#     print(\"read read_parquet:\", row[1].path)\n    x = pd.read_parquet(os.path.join(DATA_DIR, row[1].path), columns=feat_columns)\n#     display(data.head())\n    n_frames = int(x.shape[0] / ROWS_PER_FRAME)\n    x = x.values.reshape(n_frames, ROWS_PER_FRAME, len(feat_columns))\n#     print(data.shape) # output, e.g. (23, 543, 3)\n\n    lips_x = x[:,lips,:].reshape(-1, N_LIPS*3)\n    lefth_x = x[:,468:489,:].reshape(-1, N_HAND*3)\n    pose_x = x[:,489:522,:].reshape(-1, N_POSE*3)\n    righth_x = x[:,522:,:].reshape(-1, N_HAND*3)\n    \n#     lips_x = lips_x[np.any(np.isnan(lips_x), axis=1) == False, :]\n    lefth_x = lefth_x[np.any(np.isnan(lefth_x), axis=1) == False, :]\n#     pose_x = pose_x[np.any(np.isnan(pose_x), axis=1) == False, :]\n    righth_x = righth_x[np.any(np.isnan(righth_x), axis=1) == False, :]\n    \n    lips_mean = np.mean(lips_x, axis=0)\n    lefth_mean = np.mean(lefth_x, axis=0)\n    pose_mean = np.mean(pose_x, axis=0)\n    righth_mean = np.mean(righth_x, axis=0)\n\n    lips_std = np.std(lips_x, axis=0)\n    lefth_std = np.std(lefth_x, axis=0)\n    pose_std = np.std(pose_x, axis=0)\n    righth_std = np.std(righth_x, axis=0)\n\n    result = np.concatenate((lips_std, lefth_std, pose_std, righth_std, lips_mean, lefth_mean, pose_mean, righth_mean))\n    result = np.where(np.isnan(result), np.full(result.shape, 0.0, dtype=np.float32), result)\n\n#     print(lips_mean.shape)\n#     print(lefth_x.shape)\n#     print(pose_x.shape)\n#     print(righth_x.shape)\n#     print(result.shape)\n\n    return result, label_map.get(row[1].sign)\n\n\ndef convert_and_save_data(subset):\n    np_features = np.zeros((subset.shape[0], FEATURE_SIZE))\n    np_labels = np.zeros(subset.shape[0])\n\n    with mp.Pool() as pool:\n        results = pool.imap(create_features, subset.iterrows(), chunksize=250)\n        for i, (x, y) in tqdm(enumerate(results), total=subset.shape[0]):\n            np_features[i,:] = x\n            np_labels[i] = y\n    \n#     # loop over dataframe rows\n#     for index, row in subset.iterrows():\n#         np_features[index] = create_features(row)\n#         np_labels[index] = label_map.get(row['sign'])\n#     #     print(f\"shape: {np_features[index].shape}, label: {row['sign']}, sign: {np_labels[index]}\")\n\n        np.save(\"features.npy\", np_features)\n        np.save(\"labels.npy\", np_labels)\n        \n        \nsubset = train_df[:100]\nconvert_and_save_data(subset)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:26.351520Z","iopub.execute_input":"2023-04-19T05:47:26.351846Z","iopub.status.idle":"2023-04-19T05:47:28.666409Z","shell.execute_reply.started":"2023-04-19T05:47:26.351816Z","shell.execute_reply":"2023-04-19T05:47:28.665252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_x = np.load(\"/kaggle/working/features.npy\")\n# train_y = np.load(\"/kaggle/working/labels.npy\")\n\n# print(train_x.shape)\n# print(train_y.shape)\n\n# flat_frame_len = train_x.shape[1]","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:28.675087Z","iopub.execute_input":"2023-04-19T05:47:28.675581Z","iopub.status.idle":"2023-04-19T05:47:28.682725Z","shell.execute_reply.started":"2023-04-19T05:47:28.675531Z","shell.execute_reply":"2023-04-19T05:47:28.681272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_x = np.load(\"/kaggle/input/gislr-feature-data/feature_data.npy\").astype(np.float32)\ntrain_y = np.load(\"/kaggle/input/gislr-feature-data/feature_labels.npy\").astype(np.uint8)\n\nprint(train_x.shape)\nprint(train_y.shape)\n\nflat_frame_len = train_x.shape[1]","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:47:28.684134Z","iopub.execute_input":"2023-04-19T05:47:28.684510Z","iopub.status.idle":"2023-04-19T05:48:03.392600Z","shell.execute_reply.started":"2023-04-19T05:47:28.684465Z","shell.execute_reply":"2023-04-19T05:48:03.391098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_x = np.load(\"/kaggle/input/isolated-sign-language-aggregation-preparation/feature_data.npy\").astype(np.float32)\n# train_y = np.load(\"/kaggle/input/isolated-sign-language-aggregation-preparation/feature_labels.npy\").astype(np.uint8)\n\n# print(train_x.shape)\n# print(train_y.shape)\n\n# flat_frame_len = train_x.shape[1]","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:48:03.393873Z","iopub.execute_input":"2023-04-19T05:48:03.394208Z","iopub.status.idle":"2023-04-19T05:48:03.398891Z","shell.execute_reply.started":"2023-04-19T05:48:03.394177Z","shell.execute_reply":"2023-04-19T05:48:03.397528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 64\n\nN_TOTAL = train_x.shape[0]\nVAL_PCT = 0.1\nN_VAL   = int(N_TOTAL*VAL_PCT)\nN_TRAIN = N_TOTAL-N_VAL\n\nrandom_idxs = random.sample(range(N_TOTAL), N_TOTAL)\ntrain_idxs, val_idxs = np.array(random_idxs[:N_TRAIN]), np.array(random_idxs[N_TRAIN:])\n\nval_x, val_y = train_x[val_idxs], train_y[val_idxs]\ntrain_x, train_y = train_x[train_idxs], train_y[train_idxs]","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:48:03.400469Z","iopub.execute_input":"2023-04-19T05:48:03.401364Z","iopub.status.idle":"2023-04-19T05:48:03.931728Z","shell.execute_reply.started":"2023-04-19T05:48:03.401326Z","shell.execute_reply":"2023-04-19T05:48:03.930256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"def fc_block(inputs, output_channels, dropout=0.2):\n    x = tf.keras.layers.Dense(output_channels)(inputs)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Activation(\"gelu\")(x)\n    x = tf.keras.layers.Dropout(dropout)(x)\n    return x\n\ndef get_model(n_labels=250, init_fc=512, n_blocks=2, _dropout_1=0.2, _dropout_2=0.6, flat_frame_len=flat_frame_len):\n    _inputs = tf.keras.layers.Input(shape=(flat_frame_len,))\n    x = _inputs\n    \n    # Define layers\n    for i in range(n_blocks):\n        x = fc_block(\n            x, output_channels=init_fc//(2**i), \n            dropout=_dropout_1 if (1+i)!=n_blocks else _dropout_2\n        )\n    \n    # Define output layer\n    _outputs = tf.keras.layers.Dense(n_labels, activation=\"softmax\")(x)\n    \n    # Build the model\n    model = tf.keras.models.Model(inputs=_inputs, outputs=_outputs)\n    return model\n\nmodel = get_model()\nmodel.compile(tf.keras.optimizers.Adam(0.000333), \"sparse_categorical_crossentropy\", metrics=\"acc\")\nmodel.summary()\n\ntf.keras.utils.plot_model(model)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:48:03.933368Z","iopub.execute_input":"2023-04-19T05:48:03.933810Z","iopub.status.idle":"2023-04-19T05:48:04.472285Z","shell.execute_reply.started":"2023-04-19T05:48:03.933769Z","shell.execute_reply":"2023-04-19T05:48:04.470685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cb_list = [\n    tf.keras.callbacks.EarlyStopping(patience=5, restore_best_weights=True, verbose=1),\n    tf.keras.callbacks.ReduceLROnPlateau(patience=2, factor=0.8, verbose=1)\n]\nhistory = model.fit(train_x, train_y, validation_data=(val_x, val_y), epochs=100, callbacks=cb_list, batch_size=BATCH_SIZE)\nmodel.save(\"./models/asl_model\")","metadata":{"execution":{"iopub.status.busy":"2023-04-19T05:48:04.474936Z","iopub.execute_input":"2023-04-19T05:48:04.475374Z","iopub.status.idle":"2023-04-19T06:42:58.723924Z","shell.execute_reply.started":"2023-04-19T05:48:04.475330Z","shell.execute_reply":"2023-04-19T06:42:58.722855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.evaluate(val_x, val_y)\nfor x,y in zip(val_x[:10], val_y[:10]):\n    print(f\"PRED: {decoder(np.argmax(model.predict(tf.expand_dims(x, axis=0), verbose=0), axis=-1)[0]):<20} – GT: {decoder(y)}\")","metadata":{"execution":{"iopub.status.busy":"2023-04-19T06:42:58.728818Z","iopub.execute_input":"2023-04-19T06:42:58.729749Z","iopub.status.idle":"2023-04-19T06:43:00.870302Z","shell.execute_reply.started":"2023-04-19T06:42:58.729690Z","shell.execute_reply":"2023-04-19T06:43:00.869299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Convert to tflite","metadata":{}},{"cell_type":"code","source":"# Convert the model.\nconverter = tf.lite.TFLiteConverter.from_keras_model(model)\ntflite_model = converter.convert()\n\n# Save the model.\nwith open('/kaggle/working/models/model.tflite', 'wb') as f:\n    f.write(tflite_model)\n    \n!zip submission.zip /kaggle/working/models/model.tflite","metadata":{"execution":{"iopub.status.busy":"2023-04-19T06:43:00.871660Z","iopub.execute_input":"2023-04-19T06:43:00.871987Z","iopub.status.idle":"2023-04-19T06:43:05.213414Z","shell.execute_reply.started":"2023-04-19T06:43:00.871954Z","shell.execute_reply":"2023-04-19T06:43:05.212015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ROWS_PER_FRAME = 543  # number of landmarks per frame\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T06:43:05.215560Z","iopub.execute_input":"2023-04-19T06:43:05.215951Z","iopub.status.idle":"2023-04-19T06:43:05.222617Z","shell.execute_reply.started":"2023-04-19T06:43:05.215910Z","shell.execute_reply":"2023-04-19T06:43:05.221653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class PrepInputs(tf.keras.layers.Layer):\n    def __init__(self, face_idx_range=(0, 468), lh_idx_range=(468, 489), \n                 pose_idx_range=(489, 522), rh_idx_range=(522, 543)):\n        super(PrepInputs, self).__init__()\n        self.idx_ranges = [face_idx_range, lh_idx_range, pose_idx_range, rh_idx_range]\n        self.flat_feat_lens = [3*(_range[1]-_range[0]) for _range in self.idx_ranges]\n    \n    def call(self, x_in):\n        \n        # Split the single vector into 4\n        xs = [x_in[:, _range[0]:_range[1], :] for _range in self.idx_ranges]\n        \n        # Reshape based on specific number of keypoints\n        xs = [tf.reshape(_x, (-1, flat_feat_len)) for _x, flat_feat_len in zip(xs, self.flat_feat_lens)]\n        \n        # Drop empty rows - Empty rows are present in \n        #   --> pose, lh, rh\n        #   --> so we don't have to for face\n        xs[1:] = [\n            tf.boolean_mask(_x, tf.reduce_all(tf.logical_not(tf.math.is_nan(_x)), axis=1), axis=0)\n            for _x in xs[1:]\n        ]\n        \n        # Get means and stds\n        x_means = [tf.math.reduce_mean(_x, axis=0) for _x in xs]\n        x_stds  = [tf.math.reduce_std(_x,  axis=0) for _x in xs]\n        \n        x_out = tf.concat([*x_means, *x_stds], axis=0)\n        x_out = tf.where(tf.math.is_finite(x_out), x_out, tf.zeros_like(x_out))\n        return tf.expand_dims(x_out, axis=0)\n\n# /kaggle/input/asl-signs/train_landmark_files/26734/1000035562.parquet\nPrepInputs()(load_relevant_data_subset(os.path.join(DATA_DIR, train_df.path[0])))\n\n# train_x = np.load(\"/kaggle/working/features.npy\")\n# train_y = np.load(\"/kaggle/working/labels.npy\")","metadata":{"execution":{"iopub.status.busy":"2023-04-19T06:43:05.224437Z","iopub.execute_input":"2023-04-19T06:43:05.224782Z","iopub.status.idle":"2023-04-19T06:43:05.314439Z","shell.execute_reply.started":"2023-04-19T06:43:05.224752Z","shell.execute_reply":"2023-04-19T06:43:05.313349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TFLiteModel(tf.Module):\n    \"\"\"\n    TensorFlow Lite model that takes input tensors and applies:\n        – a preprocessing model\n        – the ISLR model \n    \"\"\"\n\n    def __init__(self, islr_model):\n        \"\"\"\n        Initializes the TFLiteModel with the specified preprocessing model and ISLR model.\n        \"\"\"\n        super(TFLiteModel, self).__init__()\n\n        # Load the feature generation and main models\n        self.prep_inputs = PrepInputs()\n        self.islr_model   = islr_model\n    \n    @tf.function(input_signature=[tf.TensorSpec(shape=[None, 543, 3], dtype=tf.float32, name='inputs')])\n    def __call__(self, inputs):\n        \"\"\"\n        Applies the feature generation model and main model to the input tensors.\n\n        Args:\n            inputs: Input tensor with shape [batch_size, 543, 3].\n\n        Returns:\n            A dictionary with a single key 'outputs' and corresponding output tensor.\n        \"\"\"\n        x = self.prep_inputs(tf.cast(inputs, dtype=tf.float32))\n        outputs = self.islr_model(x)[0, :]\n\n        # Return a dictionary with the output tensor\n        return {'outputs': outputs}\n\ntflite_keras_model = TFLiteModel(islr_model=model)\ndemo_output = tflite_keras_model(load_relevant_data_subset(os.path.join(DATA_DIR, train_df.path[0])))[\"outputs\"]\ndecoder(np.argmax(demo_output.numpy(), axis=-1))","metadata":{"execution":{"iopub.status.busy":"2023-04-19T06:43:05.316136Z","iopub.execute_input":"2023-04-19T06:43:05.316895Z","iopub.status.idle":"2023-04-19T06:43:05.788185Z","shell.execute_reply.started":"2023-04-19T06:43:05.316849Z","shell.execute_reply":"2023-04-19T06:43:05.786951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras_model_converter = tf.lite.TFLiteConverter.from_keras_model(tflite_keras_model)\ntflite_model = keras_model_converter.convert()\nwith open('/kaggle/working/models/model.tflite', 'wb') as f:\n    f.write(tflite_model)\n!zip submission.zip /kaggle/working/models/model.tflite\n\n!pip install tflite-runtime\nimport tflite_runtime.interpreter as tflite\n\ninterpreter = tflite.Interpreter(\"/kaggle/working/models/model.tflite\")\nfound_signatures = list(interpreter.get_signature_list().keys())\n# if REQUIRED_SIGNATURE not in found_signatures:\n#     raise KernelEvalException('Required input signature not found.')\nprediction_fn = interpreter.get_signature_runner(\"serving_default\")\n\noutput = prediction_fn(inputs=load_relevant_data_subset(os.path.join(DATA_DIR, train_df.path[0])))\nsign = np.argmax(output[\"outputs\"])\n\nprint(\"PRED : \", decoder(sign))\nprint(\"GT   : \", train_df.sign[0])","metadata":{"execution":{"iopub.status.busy":"2023-04-19T06:46:33.015517Z","iopub.execute_input":"2023-04-19T06:46:33.016378Z","iopub.status.idle":"2023-04-19T06:46:50.440538Z","shell.execute_reply.started":"2023-04-19T06:46:33.016335Z","shell.execute_reply":"2023-04-19T06:46:50.438791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# # Define your model architecture\n# model = keras.Sequential([\n#     keras.layers.Dense(64, activation='relu', input_shape=(...)),\n#     keras.layers.Dense(10, activation='softmax')\n# ])\n\n# # Compile your model\n# model.compile(optimizer='adam',\n#               loss='categorical_crossentropy',\n#               metrics=['accuracy'])\n\n# # Train your model\n# history = model.fit(preprocessed_input_data, labels, epochs=10,\n#                     validation_data=(val_preprocessed_input_data, val_labels))\n\n# # Evaluate your model\n# test_loss, test_acc = model.evaluate(test_preprocessed_input_data, test_labels)\n\n# # Fine-tune your model","metadata":{"execution":{"iopub.status.busy":"2023-04-19T06:43:24.021410Z","iopub.status.idle":"2023-04-19T06:43:24.021957Z","shell.execute_reply.started":"2023-04-19T06:43:24.021712Z","shell.execute_reply":"2023-04-19T06:43:24.021742Z"},"trusted":true},"execution_count":null,"outputs":[]}]}