{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# [W&B] American Sign Language Detection","metadata":{}},{"cell_type":"markdown","source":"![image.png](attachment:7e068ada-ac98-4392-af47-eecf8c4412de.png)","metadata":{},"attachments":{"7e068ada-ac98-4392-af47-eecf8c4412de.png":{"image/png":"iVBORw0KGgoAAAANSUhEUgAAAioAAACCCAYAAABo4OMZAAAgAElEQVR4nO3dz5Pb5pkn8Acrl1J2WRV0vGWXu2pWYO2UprK1k2ZLf4BAnSM32zd7Dw3mIM+NbDvnkMw5Y7KP8lSF7BzsW9iKcxbQd0dkO9mZkWqmAW3K7RlvbNJSIo80Ur97wILCjxfAix8kwNb3U4WyxSaBF+QL4MH7vnheiTHGCAAAAKB82H8pugQAAAAAYRCoAAAAQGkhUAEAAIDSQqACAAAApYVABQAAAEoLgQoAAACUFgIVAAAAKC0EKgAAAFBaCFQAAACgtBCoAAAAQGkhUAEAAIDSQqACAAAApYVABQAAAEoLgQoAAACUFgIVAAAAKC0EKgAAAFBaCFQAAACgtBCoAAAAQGkhUAEAAIDSQqACAAAApYVABQAAAEoLgQoAAACUFgIVAAAAKC0EKgAAAFBaCFQAAACgtBCoAAAAQGkhUAEAAIDSQqACAAAApYVABQAAAEoLgQoAAACUFgIVAAAAKK2Xii4AffWJvTw+ITq/TiRfJVq/QXTuQtElS+bpQ6L/+wnRv39C9Owh0SuXiF5/l+i1HxddMgAAgJUlMcZYYVv/8iOik4+Cr79yieiHHy+/PFn847tE390Lvr5+g+jNG8svDwAAwOpjxXX9PD7hBylERI/u2S0Tq+LrT/lBCpG9jw9/t9zyAAAAnBHFBSpf/zb6739eoYt73L48CgliAAAAIFJ5B9M+fVh0CcTFjad5tkL7AgAAUCLFBSoXrkT//ZVLyylHHl6OKWvcvgIAAABXcU/9XLhiL7zxG+fXid54N79tPT6xnyz69pDo3Kt2YPHme0TfezOf9b/xjt398+Qk+LeXLyFQAQAASKnYp36ePST644f2YFTHq1eIKh2i8zkFEY9PiO79XTCIOL9OdOlmfsHKkxOiu77tvHqF6K9/sXqPWgMAAJQDKzZQcTw5IXr8pR085BU4OO69F/7UzYUrdrCSp0f3nudRQYACAACQRUkClUV59pBoUot+T1VHQAEAAFBOBeZRWQaRp22e/nnx5QAAAIBUznagcn49urVkEV1NAAWYzWZkGAZZllV0UQAAcnW2AxUiO4V9GMzDAyFmsxl1u12q1Wq0vb1N+/v7RRcpYDab0d7eHtVqNapUKlSr1Wh3d7foYsELzjAMajQaVKvVqNFoIHiGzIqflHDRXn/HTh73pS9d/5s3ooMYeGFZlkW1Ws1zgj04OCDLsqjdbhdYMvsicHh4SAcHBzSZTAJ/l2W5gFIB2EajEb399tue1/b39+n27dukqmpBpYJVV+xg2kf37EeTvz20g4nvvWl3x7z+Tv65R56cPH8iR1bzH0D78Hd2rpbv7j3fl7zztcBS7O7uUr/f5/5tOp0WGgxIkhT5d03TaDAYLKk0Z59hGHR0dESz2Yym0ymtra3RxsYGVatVUhSl6OKVCmOMLl++zA2ga7Ua3b59u4BSwRnAimtR4c2c/OihHUzMDLtb5q8+yC+gOL9uL3l7+pDoftcus5uzL1//ttAZlCeTSaruAFmWaTQaeV5rtVp0dHSUeF3NZpPq9XrqMmmaRjs7O4m3mxbvROv+W5F3hqqq0mQyodlsVlgZzjqnS63f70d+z6qq0mAwKDxgyaN7RZZlqlartLGxQaqqpg7Gw44dwzC4rwOIKCZQ4QUpfl//lujJl/nnOcnb8U/jZ0d29rWAYMUZZJkU7+R7dHSUal2apmUq07IDg2q1Glq+oi9Kuq7P/38ymVCj0fBcHND1k41lWbS9vT3/TmVZpp2dnfnF++joiPr9PlmWRYZhUKVSoeFwuNRA2i+vQdQHBwfz/9c0jdrtduL6rigKtyzVajVz+eDFtfyun8cnRH94S/z9f/WB3RVURl9/SmR1xd9/6ebS0+nPZjPa398n0zTp/v37oUGCoiieZWNjw9MKQvR8fIRlWWRZFs1ms8AdlCzL83VcvHiRKpUKbW1teU547jIdHR3N1+dfh9O8vrW1tdQTnWVZtLm5GbibbjaboV1CRanVap7fs9PpFD6OZpVVKpV5XVQUhXRd516sNU3zDLDWdb2wlraDgwPPceQ/nhz+GwYi+1i0LCu0JSRpfer1evT+++97XpMkiQaDQaHBHKy0AhK+WV1vyvw4598k+tsE71+m37/Fn98nzGs/JlI6iyuPIP/FjSj9ida/rrTrqdfrdOvWLSIqfhwIkR2s7O7uzk/gOzs71OkU/9v5FR2obG9vzwO6Xq+30nfO3W7X8xubphnaojCbzWhzc3MeEKiq6mntKpJlWVSpVAKvR53qZ7MZHRwcULfbDQQ5SevUcDik/f19siyLFEUJdP0CJFTAGJW4bhK/J1/a6fXLNiD10b1kQQpR8n1fkHq9HghUbt26lTjAcJq/3fb391MFKs7Ylyz943lSFCUwRge8LMvydBes+riZ4XA4/39VVSO7PWRZJk3T5oGNM26oLHU3KWd/VFX1dH0R2YGKoijCLSKapnFbbwDSWn4elaQX97SfWTSRrLd+T77MvxwpbG1tBV5zX3BE8bqQ0qxnMpl47kxhNfgHVpfhIp2Wuw4SidVDd+vRbDZb+UCNKDxA73YTdHED5Gz5gUqap3heKuFcPGlaeEoyp5AzhsTNGXOSBC8JWprBu4eHh/P/v3r1aqLPQnF445NWVZog4/vf//4CSlI8RVECgRqv9RRgWZYfqLxyKdn7z12w85Fk9fSh/QjxzLC7krKKS8/Pk8d+5ITXZ+yMERERdeJKsh6i560wsiyjRWWFjMfjoouwMCKBy1lqUfLjnR/SpCYAyMPyA5WkaevlHO6wv/rEftLoX39qL3+4HsxUm0bSffmv17NvMye87p8kd0zu9/oHULr7+uO4W2AQpKyWs3Sh9rcwihwL7m7Osoytyou/tYgxdia6tmA1LX8w7WvX7RwpIgNLz68Trb+XbXtffUL0x78Pvn7ykd0ikuXR5/UbRLNDsTE0L18q1dxC1WqVZFn2nHwODg6Es5o63T7VapU0TaNWqzX/mxN8iAQe7m6fLE8GTCYTT+6LPFtnnEc4szzR4oyBcL5v96PgcduezWa55G/xP4rqPAKe5gLrrMst64XaXz7nd0xbxiQURfEcD5PJJLIOD4dDTzCDwaN8k8lknm4gLadOOL+NO31BXsLqXp43T7xt5LkfRR4/C8eK8PgLxj6/zthnV8KXz68z9uhu9m1FbWesZl//X+7G78v/foexxyfZt5UzVVUZEXkW0zRjP2ea5vz9nU6HTafTwHparZZQGXZ2dhJt203XddZsNpksy4HtO4umaYnX615/vV6frz/pesbjcWz5FEVh/X7fs+7pdMr6/b7n9wnj/w07nQ53P1RV5ZZDlmXuZ3jrGAwGrNPpME3TWLVaDaxLVdXQJW7dvLroX/dwOIwtZxatViuwTR7TNJmiKPP3NZvNhZYrDd53mES/3/d8VpIkNhqNhD47nU7ZaDSa/6aVSiVx+UWObVmWMx3fjDE2GAwWXvfi6reiKJnPU2U4fhbotJhAxfHFzeBFfqwyZrbzubA//iI6gPjsCmP/kdN2zA4/EPriJmNPH2TfxgL4T0ZExPr9fuznBoPB/P3j8ZgxFrxgKooiVAbngif6fsbsE6H/ohK3iFyMnXV3Oh3uhTjJiaTT6XBPrM6Fm3ey6nQ6oSecMFGByng8jj2BuQO6KO4Lc5ol7LsWLZ/7exK9YCal63rs8eAPUnZ2dhZSlqyyBipbW1uBQCWu/ocFF5ubm8Lb/eabbxIf25VKJfFF/vj4OHHdq1Qqiepe0m04Ny1l2oeSKDhQcfzlLmMPPrP/m+dF/emD+EAl7+09+Oz5vpScu2XEWeLufhl7fnF0Bxe8oEfXdeHtx10o3Z/xXzRVVWW6rrPpdMoYsy/Qg8Eg8L6obYhc1J31x/GfaBVFCXwX0+mUtdvtTBd6xsIDFd7vEbdE3XHlHahMp9PAOtvttud3dO7M3a1uonU0Lf+23PVY13XPRTiqJUXXdaElS2tAlCR1yI93XohqIfW3/vmXWq0mtN3j42NPnZAkKfAdhR3blUpF+Pj0b0eWZdZut9l4PA6cQ/z71Wg0hLYxGAwCAZumaZ76res6t76J3FSF7QPv+KnX66n2oURKEqgs0t0b4UHKP98ounSF8x/wsixHvj8suOB1/8QdcO6WGZEonxektNvtyPf7W0bCyqRpWupgIWyfnCAl6mLU6/UC2+n3+2wwGHiWMLxAxd+ao6oq63Q683U1m01u4FGtVoX2kbFgi1GSFjHGggFB3O/vLq9oUJsGL4CSZdkTfPICTzfehT5sWVRzfJZAxX9hiwoCTk9PY/dRJFA5PT0N1OWoLqPj42NPICBJklBrxOnpqWf/JEmatwqHvd+9nd3d3dht/PrXv/bsx9raWmR9uXPnTmBfot5/enrqOX6S7gMClTIKGw8zVks5bmTZeM2sUQeJ+0Lsf1/S7h9387LI3VCSIMVhmmbgzoYXODh3T7wunyQnen8ZRS5EIuNMRD/rHzsR9r3ygr6w74YnS6Div5CLtJC468oiAxXG+F1ATsDSbrdj6yovQA5b36Ka4dPUX95Yh83Nzcg64Vz4VVUNbXUTCVR++ctfBj4XFRScnp4Gzl0i27l9+7bnM3EX7dPTU89vGXds8gKoqCAirFxxQZr7vdvb27H74P5dEaiUlTOG5PPr9pLXGJgzgHdSjjoYed0+Dl53Q9RJzjmgRS5UWS6M/s+KDPTljTGJ4/8u41qnHP7vLcm+hY13iet245VXNLBiLNvv4d9fkcDD/ZlFByphY2eq1apw94KzHmcQsrvso9FoYV0+Dl7AwOt6clrY/PsrOsjaz39hFwkgeK0pca0KjAWDm7W1tdjt+IObuH30fybq/f7WmiRBgf87iNp/fytsnvtQUqfLz6NShPPrREqb6G9/Yy9Kx57sELiProXlkIjLecKbC4SXvdbZhvO4YdzjebPZLJCbJckEgf5yieR5SfM4n/tRayLxqe39OW3SZAl2ODP+Jk0B7972oqXZhvvx1jwe1Q6zt7dHlUqFDMMI1IHJZELb29vC63Ieb11bW5u/5kzQt8h9CFOr1QJLo9Ggvb29+f5qmkaDwYBM00w9uWWaY8dfJ3jZcf389Zf3yLyfaZqef3/77beR75ckiX70ox9FvsdhGIYnt44kSfSzn/1M6LOSJHnOA4wx+tWvfhV4H2MskL/o/v37sesW3YeyejECFQjlPGfv5g4i3NwHIS9hHC/vQFjQ476o89bl3677BCTLcqIp4/0XBZETWpr06LyTbVppA5W4yfTcePktlhGo+Pl/X56trS1ijBFjbCGzQ89mM6rVatRqtWg2m1G1WqXxeByoZ4Zh0O7ubqJ1O3Pn5J37I2/ODMr7+/t069at1HUhab33X9DX1tboww8/jP0cLyCKK7M7aCSyb1rijrVGoxFb9xhjgZuyJMciUbqkg0Rix4/IPpRZ8YHKzCC6956dOfaf3rWTwa2qLz+y9+MPb9n7tCL7wku0FjXhoCzLocnZ/K8bhsE9iJz1iyRV8p8Akp7s/XPSEBVzQS6jIu7seYFjrVbj/k7LYFkWbW5uzutkvV4nXddJURTq9/uB+tbv9xNN0ucE5UVnXtZ1PbCMRiPqdDpUr9fnCe8MwyBN06hSqVCj0VjKsdJoNGg6ndJ4PKbj4+NMyR/DSJIUqHvT6ZQuX76ceR9ns5nnPOVvIRHhLwOvTJIk0cbGhue16XRK165dO9vntMJ6nRhj7OQm/2mcf/mg0GKlYnHyqHx2xd7HkhNJ2OZ+T71eT7Qu/2h80XU54soWJ+mAYcaCT++IHCr+7YiOpRiPx4FtiY6F8PftJx2/kfbzWcaoTKfT0EReqqouNc+D/ykfRVEC333YwGPRRHnO+0UGVeaF992K4D36K7qvjNnjIfxP0Ik+npzEdDoNjFERGdfiH7Tqr/siY7tE1itSFrdvvvkm8L1LkhT63kXsQ4mdLj+FvuPxiZ3Gnmdm2K0RJUo5H+nrT4n+9Cn/bycfEf3gerrZlpfE6f5x39EeHBxQr9fz/NsRdbfjtJC4W2QODg6o2WzO/50kbT7vLtswDGo0GpGfI3o+pmZZc5T473REm279++ikcz+rZFmmdrvN7UIxDIMMw5iPUWg2mwvtLmm1Wp470cFgEPjunXE/m5ubnrrU6XRIUZTIbkhnPFTZu30cmqbR1tYWXbt2zVMvnTFhRXYbGIZBh4eH8zqShqqq1Gq1qN/vB/42HA5pOBySoii0tbVFrVZLuMXRP26EMUbdbjd0jJ5b0vOULMtC+6CqKrXb7UJaTXNXWIz0bx9HJ2LLu1XlP06eJ2PLIxut2798EL0v//ZxvttbAF6rg/vOMsmjxLynf9yfcecAiLvLDHtUNOsSt900LSq81iSRu+gsrSKr2KLiEM1Auqj03/66FZdHZjQaccsXVjb3o/HLTl/OK2cSYXf4cXfrebeoOJmieblt/E/YiLZi+POQxNU9kX1Omk1XZInK6JtkmyL7UHIFtqg8exj996cxfxf18Hf22BH/JIivXCJ6/d3ltNrE7WsJbG1tBSL0W7du0c7ODs1mM7p16xYRic0Su7Oz45mkkMgeZ+K0qjgtKiJ3mbx+13q9nvnudBF3t84gX/ddVLfbnQ+m5PFPbkdU7F3rMvV6PdrY2KButxvZv+7cQRuGkesdov/pr7jWvXq9Tu12OzA+pdVq0cbGRuig9LhWlzJaW1ujZrPpafVijNHPf/7zpYy1MQyDut2u59hwjq96vU6qqpJpmp6WXlGSJNFgMCBFUWLHGjn1TtO0yLrHaw1ptVqZWkaj6rkkSdTr9ejixYu0t7cndPzE7UOpFRYjPfgsuhXi//wi+zbCxsDkPYbki5jtTFcjmuWlfGbM27ogemfov1N3cqUkTZvPa9lYxt1pmhYVxviZTcP69/0p2YmI9Xq9ROVc5RYVh2marNfrCaXpj8v0m4T/uxe96+TdycqyHGg9c/aniMngeN9dUryxEJIkRbaoZm1R4c1fE9Yi4E98lnRciLOOnZ0doboXNqcQb59F5kbKi2marN1uZ9qHkiswj8qFK3arBs/5daI33s22/q8/DR8D43byEdFXn2Tb1hvvEJ27wP/b+XUiudjR/qKuXr3q+bdzt+K+a/G/Jwzv6R93Hhai+MeSiYKPExKV+4kdWZbnT4w4Op0OVSoV6na7NBwOaW9vb57Hwn0n1uv1Ai1RLwJFUajVapFpmqTrOu3s7ITeiVqWFfje0phMJqnX0ev1Aq0Ks9mMtre353VzOBySZVkr2ZrikGWZ+zss6ums4XBIly9f9jwR2Ov1hPMCpVGpVGg4HJJpmjQajSJb1UzTpGvXrnHrDe97WtbYOEVRqNPpeI6fMFH7UGbFPp78338RvIifXyf6619kS8j27CHRyT+Iv//ko2zdM+cuEF26aZfd7dUrRH9zM/16l6xWq3n+7QQW7m4f0WbDsORv7qBHpPvFP0CVqNyBCtHzwZfu78qyLOp0OtRoNKjVankCNlVVaTwev5BBip+qqvMLh9M872dZFu3t7WXaTtYT9Wg0CtRfJ4hyui2IkiUmLKNlXYBHoxE1Go35utfW1kjX9aUeE/V6nUajUWTdM00zUPckSaKLFy8G3usfYLsMIscPbx/KrthA5fy6Haz88GP7v5du2pljXw5paRE1M4ienIi//9lDoulh/PuivHLJLvulm3YW3B9+bAcpK5QBlxdcuPuok+Q24OVHOTg4mI9PqVarQkEP7wmYNP3Sy+aMu9A0jXRdp06nQ5qmzZdms0mDwYCm0ynpur4ST4Qsk5MlNSxDKu+Jh6ySXFhkWabRaMRNmFer1Va+NcUh2nqQhWma9P7778//7Yy/KOqYUBSFNE2jO3fucAMl58bNzX9DxRgLZKpeJmcfjo+PucePyNNIZVJ8wjci+yIvq3Z3UB4e3Uv+me/u5rPtC1eIXrse3q1VYrzgwt3MK9rt4+BlqXVOfEmacnmpstM+nrgMTheP8xiu85jgYDCYL/1+nzRNO9OPIeel0+kELhgi2YWj8L73pAGw03IW9htGDaJeBWFTOeQdQPCSQiY91yyCkyE36pzo4E1FcnBwUHgXiyRJ1G63AwFzlmk6ilCOQCVvz/6c4jPlfzJnGcICiDR5IKLuJpNkbeS15CTJDOqW9QIXxzCMeXN/0hTaZSD63WQ5ya2trZEkSSRJkvB63Hl48sBrqTMMI/H4i6hgxd2VUTRJkhJ/hjfXjMhTf0kwxujzzz8PvL6I44YxNq93P/jBD4Q+I0mSUN3jTUUynU5Td7GE1Zu0+8A7F5elboo4m4FKmu4W//iSF1TYnUyalNZRE4slaVHhDa50jwMQ5aRqz2MwZhh3V1kRrSXLGr+T5ftzfy9ZBmZmuZjxLixE6YKLarXK7SJIOolhmZimGeheC7vgZcX7vkXqcZa6nuSGxT/+hHdch01A2O12E9dx0zRpc3OTfvKTn0S+bzabCa/bf6zw5vkqs7MZqKTpQsqr22nFhd0xJZ23wr0+kdeiOJkY/TqdjnCw4owdmEwm3GbaPPhPHGUf9EuUvoz+7y9tU7LouBB/OfN4CoTXdz+ZTBLPb9PtdkPHzIhmUS6TsCdDms2mUKCS5Lvjzb/DOBP8+XW7Xbp27Vrg9SR1UHQMiT8YCLtpU1U1UC8ZY/T2228LBxS6rs/n7eEN0CXyBh2i+7CI42epinw4eqHu3ojPoeIsn18vurSl4s8JkDVHBvme5ffP/SOqWq0mzrzoZLZ0cmaI5uHw5wkROVR4+6ppGhuNRkzX9dglzVwwYflqRPlzL4h+npd9mPcbTKfTwPft3qZo3hb/fuY1H1BYdk9FUdhwOAzNGWKaZiBjqqIorNlscteXdH6qLHgZksPmjXHTdZ1b75P8Tqenp4HfKi6PCm/+HUmSuOeJ27dvz88DInMSueve6empJ3fO7u5u7P74M/TG5Uc5Pj4OncdK07TQY1zXdU+23LCstKenp57yNBqN3PehhE6LDVT+84GdcO3uDcZ+f52xf3yHMbOdT4r7x18wNlbjg5SxytjjnFPqrzh/srOkScT8/CeutJOzhU0O5yyyLDNVVecLL+121Lan0ykbj8es3+8HyuycBEejUehBHjXZXpIlLmW8aZpsNBqxfr8f2J4sy6zf7zNd10MvstPplOm6nvrzzjp4F3cngDBNk/X7faYoCpNl2fOdub9bWZYjk3Tpuh74Ldrtduj704hLp66qKqvX60zTNG69cvbd2cew4MdZT6fTYYPBIPY7FuWuD81mM5Ba3l1//Yvz/rDjSpZloSSE0+l0XgbecRdVp3jBDe+YdtfVSqXCTQ63trY2vzGo1+usUqnMt+m/yFcqlcjzgTsoCguEeO7cuRN5HvCfp/zvjUrK5t+HtbW1yOMn7T6UTIGByr9/HB1I5JGZ9i937daSqJaUR3ezb+eM8V+Ess4T4b5Ly9I6w5h9Ug47EUctiqJEnpQODg4SrS9s1uder5c5UHEWTdMCJ/awE3rY4m95SDonSVTLhei6/K1Y7nmj/Cdvd0DAO9nnHaQ42u126t+p2WwGfifRuWTymINFJCNp0kWWZdZut2PvvJPOcyNJErdOHR8fC+9HrVablytqNmQiYtvb256y8lpl/XWvWq1yA/gkWaOT7I9/3+Iy//LOASLHjyRJiTNfl0RBgYpIavvPrtitK1k9fcDYn35jt9qMVTs4+ecbdtr7pw+yr9/tPx/YUwP86Tf2f1eYczDIspx5XUnT5osYDAahXUG8E27cnSsvZX7UEtVFIpoOXmTxf19JAxV/y4y/Wy/p5/3iLlL1ep373Y9Go0QB5zImVjNNUzjAkGWZNZvNyAu5SPCTtnXRLWtdk2WZKYrC6vU6azabiVp6eOnjoxZJkkLrFK+FxF9OXmvAhx9+yH1/2EV/MBgIH0civ3MU0XNB0ok3k+5DvV5fte4et1OJMcZomR6fEP3hLfH3K53lTByY1aN7RP/6U2+iufPrdgK4761O0jeHk9tAluVUT/z4OTkFqtVqrnkYLMuaP1b67bffzl+vVqu0sbFR2KCx4XBIjUaD6vW60AC/qJTui0whngfLsmh/f98zYK9arUamwnc4OXGOjo4CA/4URZn/hst8giqqTE69Eh2QHVU/L168mMuxddYYhkGHh4eJ6pPzPd+/f59kWRY69i3LoslkksvvLLJPR0dHgUG11WqVrl69mvqcWMbjZwHY8gMVq2vPwyPq5UtE/+PjxZUnD49PiP7pf/FzsZxftzPWwgvBsixqNBpkGAapqkq6rgt/djKZ0O7ubiCZXavVol6vl3dRAQBWAVv+48mPEmaA/e4e0eMvF1OWvHz1SXjCuCcnRA9/t9zyQCEmkwltbm7OA43BYJDo89Vqldt6sgqPOQMALMryA5XvUqS3TzJvTxHi9gmByplnWRZtb2/Pu2/q9XrqhEr+/B6rlEESACBvyw9Uzl1I/pmXUnxmmeL2Kc0+w0rZ29sL9KmntUoZIwEAFm35gUqayfqyzqZMRPT0oT3gdRGtG3LMBFpr5R0ICfnIkgrez9+CggGXAPAie2npW3ztx8mChTye+PnqE6KTj56PIzm/TvTGO0Svv5N93UT2bMkP7/AHCb95I93cQ7BS/MFFlu4af/rwtNMXAACcBctvUXntuniryvl1ovX3sm3v60+J/vj33sGuT07s1776JNu63ZS2vVy4Ypf71Sv2o8nrN/LbBpSWv6tnOBymClYmk4ln3ph2u42uIAB4oS3/8WQiO1C4+3fRg2TPXSD6m5vZu31+/1b4ds5dIKqKPz4KEMYwDKrVap7XNE2jXq8nnMNgb2+POp3OPMCpVqs0Ho9zLysAwAopII+K48kJ0ck/8LtLXr1CVOlk7zJ5cmIHKlH+56crmZANymd3dzcwi66iKNRqtbjJo5yEU4Zh0P7+vqcFZmdnh/r9/qonagIAyKrAQK4XdAEAAAGmSURBVMXx7P8Pcn1yYrdwXLiS31Myzx4STWrR76nqeCoHctPpdKjb7ab+vKIo1Ol0aGdnJ8dSAQCsrBIEKot2773wwburkPUWVo5lWdTv9+nw8FDoaSBZlklVVWo2m6VOlQ8AUIAXIFB5dM8OVvyZY89dsIMUPJEDCzSbzebz+PgH18qyTNVqFYNlAQDCvQCBCpHdrWR17ZaVcxfslpT/9kE++VkAAABgUV6QQAUAAABWUQGTEgIAAAAIQqACAAAApYVABQAAAEoLgQoAAACUFgIVAAAAKC0EKgAAAFBaCFQAAACgtBCoAAAAQGkhUAEAAIDSQqACAAAApYVABQAAAEoLgQoAAACUFgIVAAAAKC0EKgAAAFBaCFQAAACgtBCoAAAAQGkhUAEAAIDSQqACAAAApYVABQAAAEoLgQoAAACUFgIVAAAAKC0EKgAAAFBaCFQAAACgtBCoAAAAQGkhUAEAAIDSQqACAAAApYVABQAAAEoLgQoAAACUFgIVAAAAKC0EKgAAAFBaCFQAAACgtBCoAAAAQGm9VHQBAAAA4Exhea7r/wH/ACzrwaDH/AAAAABJRU5ErkJggg=="}}},{"cell_type":"markdown","source":"## Introduction","metadata":{}},{"cell_type":"markdown","source":"Hello and welcome all to my first Kaggle notebook in a few years. Generally I write blog posts - the most recent ones being on [CLIP](https://amaarora.github.io/posts/2023-03-11_Understanding_CLIP_part_2.html), but have recently taken an interest into Kaggle. The idea is to create clear concise high scoring notebooks to share with all.","metadata":{}},{"cell_type":"markdown","source":"I started competing in this competition about 2 weeks ago and have spent few hours every day for the past few weeks experimenting on this competition. I hadn't used Tensorflow or Keras before this competition - so want to take my time to thank all the Kagglers who have openly shared their solutions. Since, I learnt from you all, I am sharing my solution too. ","metadata":{}},{"cell_type":"markdown","source":"As part of this notebook I will also share my learnings based on the experiments so far and also possible ideas that could potentially make the solution even better. ","metadata":{}},{"cell_type":"markdown","source":"> At the time of writing, this notebook has the highest score on public leaderboard and ranks $48/709$. ","metadata":{}},{"cell_type":"markdown","source":"If you'd like to team up and work in the ideas that I have shared in this notebook, feel free to reach out to me. :) ","metadata":{}},{"cell_type":"markdown","source":"## Credits","metadata":{}},{"cell_type":"markdown","source":"1. Thank you @markwijkhuizen for your highest scoring single Transformer model [notebook [LB 0.67]](https://www.kaggle.com/code/markwijkhuizen/gislr-tf-data-processing-transformer-training). \n2. Thank you @roberthatch for sharing [how to create features](https://www.kaggle.com/code/roberthatch/gislr-lb-0-63-on-the-shoulders) that have time information in them. IMHO, you have uplifted this competition as many other high scoring notebooks have been based on your feature set after. I myself tried over 60 different experiments on top of the notebook that you shared. ","metadata":{}},{"cell_type":"markdown","source":"## Experiment Tracking using Weights and Biases","metadata":{}},{"cell_type":"markdown","source":"I used [Weights and Biases](https://wandb.ai/) for experiment tracking. I ran a total of **83 different experiments** which I tracked using W&B. As part of this notebook I will also share my learnings from those experiments. \n\nWeights & Biases is the machine learning platform for developers to build better models faster. Use W&B's lightweight, interoperable tools to quickly track experiments, version and iterate on datasets, evaluate model performance, reproduce models, visualize results and spot regressions, and share findings with colleagues.\n\n> For a quickstart on W&B, refer **[HERE](http://wandb.me/aman)**!","metadata":{}},{"cell_type":"markdown","source":"![img](https://raw.githubusercontent.com/amaarora/amaarora.github.io/master/images/kaggle_nb_01.png)","metadata":{}},{"cell_type":"markdown","source":"To access all my experiments - refer [here](https://wandb.ai/amanarora/asl-sings?workspace=user-amanarora).\n\nAs part of these experiments I tried a wide range of learning rates, model architectures, dataset features. I hope that the information I share here will be helpful to you in creating a winning solution.  ","metadata":{}},{"cell_type":"markdown","source":"Now, with credits and introductions out of the way, let's get started.","metadata":{}},{"cell_type":"markdown","source":"## Imports","metadata":{}},{"cell_type":"code","source":"import os, json, random, math, scipy, wandb\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import StratifiedGroupKFold \nfrom types import SimpleNamespace\nfrom pathlib import Path","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To be able to use trained model weights, please create a wandb API key and paste below:","metadata":{}},{"cell_type":"code","source":"try:\n    from kaggle_secrets import UserSecretsClient\n    user_secrets = UserSecretsClient()\n    secret_value_0 = user_secrets.get_secret(\"wandb-api-key\")\n    wandb.login(key=secret_value_0)\nexcept:\n    wandb.login()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:06:01.529638Z","iopub.execute_input":"2023-03-26T14:06:01.530587Z","iopub.status.idle":"2023-03-26T14:06:16.785275Z","shell.execute_reply.started":"2023-03-26T14:06:01.530537Z","shell.execute_reply":"2023-03-26T14:06:16.784119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utils","metadata":{}},{"cell_type":"markdown","source":"It is important to be able to test the models, since as part of this competition we only submit a `submission.zip` file that contains `model.tflite`, it is important to check that the model is working before submitting to the competition. ","metadata":{}},{"cell_type":"code","source":"# Source: https://www.kaggle.com/competitions/asl-signs/overview/evaluation\nROWS_PER_FRAME = 543\n\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:06:30.710623Z","iopub.execute_input":"2023-03-26T14:06:30.711714Z","iopub.status.idle":"2023-03-26T14:06:30.719010Z","shell.execute_reply.started":"2023-03-26T14:06:30.711661Z","shell.execute_reply":"2023-03-26T14:06:30.717595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above code from the evaluation page showcases how the test data is loaded in the competition. So if it works on our model, we can be rest assured that our submission won't fail (unless it times out..)","metadata":{}},{"cell_type":"markdown","source":"# Model-1 Transformer","metadata":{}},{"cell_type":"markdown","source":"First, we get started with the transformer model from the amazing [GISLR TF Data Processing & Transformer Training](https://www.kaggle.com/code/markwijkhuizen/gislr-tf-data-processing-transformer-training) notebook by [Mark Wijkhuizen](https://www.kaggle.com/markwijkhuizen). ","metadata":{}},{"cell_type":"markdown","source":"The only difference between our version and the original version is that we trained the model on the whole dataset instead of just the training dataset as in the original notebook. *The single model trained on complete dataset is able to achieve a score of 0.68 on the public leaderboard.*","metadata":{}},{"cell_type":"markdown","source":"## Config and setup","metadata":{}},{"cell_type":"code","source":"cfg = SimpleNamespace()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:20.822014Z","iopub.execute_input":"2023-03-26T14:07:20.823134Z","iopub.status.idle":"2023-03-26T14:07:20.828007Z","shell.execute_reply.started":"2023-03-26T14:07:20.823081Z","shell.execute_reply":"2023-03-26T14:07:20.826872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"iskaggle = os.environ.get('KAGGLE_KERNEL_RUN_TYPE', '')","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:21.000244Z","iopub.execute_input":"2023-03-26T14:07:21.000639Z","iopub.status.idle":"2023-03-26T14:07:21.005902Z","shell.execute_reply.started":"2023-03-26T14:07:21.000604Z","shell.execute_reply":"2023-03-26T14:07:21.004564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_DIR         = Path('../data/') if not iskaggle else Path('/kaggle/input/asl-signs/')\nTRAIN_CSV_PATH   = DATA_DIR/'train.csv'\nLANDMARK_DIR     = DATA_DIR/'train_landmark_files'\nLABEL_MAP_PATH   = DATA_DIR/'sign_to_prediction_index_map.json'","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:21.465702Z","iopub.execute_input":"2023-03-26T14:07:21.466486Z","iopub.status.idle":"2023-03-26T14:07:21.472698Z","shell.execute_reply.started":"2023-03-26T14:07:21.466449Z","shell.execute_reply":"2023-03-26T14:07:21.471575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg.PREPROCESS_DATA = False\ncfg.TRAIN_MODEL = False\ncfg.N_ROWS = 543\ncfg.N_DIMS = 3\ncfg.DIM_NAMES = ['x', 'y', 'z']\ncfg.SEED = 42\ncfg.NUM_CLASSES = 250\ncfg.IS_INTERACTIVE = True\ncfg.VERBOSE = 2\ncfg.INPUT_SIZE = 32\ncfg.BATCH_ALL_SIGNS_N = 4\ncfg.BATCH_SIZE = 256\ncfg.N_EPOCHS = 100\ncfg.LR_MAX = 1e-3\ncfg.N_WARMUP_EPOCHS = 0\ncfg.WD_RATIO = 0.05\ncfg.MASK_VAL = 4237","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:23.782485Z","iopub.execute_input":"2023-03-26T14:07:23.783190Z","iopub.status.idle":"2023-03-26T14:07:23.791632Z","shell.execute_reply.started":"2023-03-26T14:07:23.783151Z","shell.execute_reply":"2023-03-26T14:07:23.790478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read Training Data\ntrain = pd.read_csv(TRAIN_CSV_PATH)\nN_SAMPLES = len(train)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:25.624138Z","iopub.execute_input":"2023-03-26T14:07:25.624534Z","iopub.status.idle":"2023-03-26T14:07:25.806723Z","shell.execute_reply.started":"2023-03-26T14:07:25.624479Z","shell.execute_reply":"2023-03-26T14:07:25.805712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Below, instead of updating the path on the train file, we create a symlink instead. ","metadata":{}},{"cell_type":"code","source":"# Get complete file path to file\ndef get_file_path(path):\n    return f'/kaggle/input/asl-signs/{path}'\n\n!ln -s {LANDMARK_DIR} ./train_landmark_files\ntrain['file_path'] = train['path'].values\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:29.620134Z","iopub.execute_input":"2023-03-26T14:07:29.620545Z","iopub.status.idle":"2023-03-26T14:07:30.766436Z","shell.execute_reply.started":"2023-03-26T14:07:29.620485Z","shell.execute_reply":"2023-03-26T14:07:30.765120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next, as in the original notebook, we convert the sign to a category and convert sign to codes and assign it as `sign_org`. ","metadata":{}},{"cell_type":"code","source":"train['sign_ord'] = train['sign'].astype('category').cat.codes\nSIGN2ORD = train[['sign', 'sign_ord']].set_index('sign').squeeze().to_dict()\nORD2SIGN = train[['sign_ord', 'sign']].set_index('sign_ord').squeeze().to_dict()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:31.380076Z","iopub.execute_input":"2023-03-26T14:07:31.381293Z","iopub.status.idle":"2023-03-26T14:07:31.552073Z","shell.execute_reply.started":"2023-03-26T14:07:31.381230Z","shell.execute_reply":"2023-03-26T14:07:31.551029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I am not sure why the author did this, could have also used the mapping provided to us in the competition. ","metadata":{}},{"cell_type":"markdown","source":"## Kfold","metadata":{}},{"cell_type":"markdown","source":"Below, we create K-folds using `StratifiedGroupKFold`, the idea  is to have different participants in train and validation. This mimics the test scenario better compared to a random train-test-split IMHO. ","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedGroupKFold","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:33.790600Z","iopub.execute_input":"2023-03-26T14:07:33.791560Z","iopub.status.idle":"2023-03-26T14:07:33.797077Z","shell.execute_reply.started":"2023-03-26T14:07:33.791480Z","shell.execute_reply":"2023-03-26T14:07:33.795711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_PARTICIPANTS = train.participant_id.nunique()\nsgkf = StratifiedGroupKFold(n_splits=7, shuffle=True, random_state=43)\ntrain['fold'] = -1\nfor i, (train_idx, val_idx) in enumerate(sgkf.split(train.index, train.sign, train.participant_id)):\n    train.loc[val_idx, 'fold'] = i\ntrain.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:33.942062Z","iopub.execute_input":"2023-03-26T14:07:33.942748Z","iopub.status.idle":"2023-03-26T14:07:34.248660Z","shell.execute_reply.started":"2023-03-26T14:07:33.942706Z","shell.execute_reply":"2023-03-26T14:07:34.247467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create indexes using fold `0` for now\ntrain_idxs = train.query(\"fold!=0\").index.values\nval_idxs = train.query(\"fold==0\").index.values","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:37.782864Z","iopub.execute_input":"2023-03-26T14:07:37.783318Z","iopub.status.idle":"2023-03-26T14:07:37.821849Z","shell.execute_reply.started":"2023-03-26T14:07:37.783278Z","shell.execute_reply":"2023-03-26T14:07:37.820565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_idxs), len(val_idxs)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:37.932628Z","iopub.execute_input":"2023-03-26T14:07:37.935755Z","iopub.status.idle":"2023-03-26T14:07:37.947656Z","shell.execute_reply.started":"2023-03-26T14:07:37.935705Z","shell.execute_reply":"2023-03-26T14:07:37.946342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Process Data Tensorflow","metadata":{}},{"cell_type":"markdown","source":"Next, we copy the `PreprocessLayer` as from the original notebook - https://www.kaggle.com/code/roberthatch/gislr-lb-0-63-on-the-shoulders with some pre-defined variables. ","metadata":{}},{"cell_type":"markdown","source":"> One thing I noted, the pose indices - they have values `np.arange(502, 512)`, is this right? Shouldn't this be `np.arange(489,522)`? Something for you to experiment. :) ","metadata":{}},{"cell_type":"code","source":"# landmark indices in original data\nLIPS_IDXS0 = np.array([\n        61, 185, 40, 39, 37, 0, 267, 269, 270, 409,\n        291, 146, 91, 181, 84, 17, 314, 405, 321, 375,\n        78, 191, 80, 81, 82, 13, 312, 311, 310, 415,\n        95, 88, 178, 87, 14, 317, 402, 318, 324, 308,\n    ])\nLEFT_HAND_IDXS0  = np.arange(468,489)\nRIGHT_HAND_IDXS0 = np.arange(522,543)\nPOSE_IDXS0       = np.arange(502, 512)\nLANDMARK_IDXS0   = np.concatenate((LIPS_IDXS0, LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0, POSE_IDXS0))\nHAND_IDXS0       = np.concatenate((LEFT_HAND_IDXS0, RIGHT_HAND_IDXS0), axis=0)\nN_COLS           = LANDMARK_IDXS0.size\nN_COLS, LANDMARK_IDXS0","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:40.648561Z","iopub.execute_input":"2023-03-26T14:07:40.649028Z","iopub.status.idle":"2023-03-26T14:07:40.670135Z","shell.execute_reply.started":"2023-03-26T14:07:40.648986Z","shell.execute_reply":"2023-03-26T14:07:40.669055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Landmark indices in processed data\nLIPS_IDXS       = np.argwhere(np.isin(LANDMARK_IDXS0, LIPS_IDXS0)).squeeze()\nLEFT_HAND_IDXS  = np.argwhere(np.isin(LANDMARK_IDXS0, LEFT_HAND_IDXS0)).squeeze()\nRIGHT_HAND_IDXS = np.argwhere(np.isin(LANDMARK_IDXS0, RIGHT_HAND_IDXS0)).squeeze()\nHAND_IDXS       = np.argwhere(np.isin(LANDMARK_IDXS0, HAND_IDXS0)).squeeze()\nPOSE_IDXS       = np.argwhere(np.isin(LANDMARK_IDXS0, POSE_IDXS0)).squeeze()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:40.809230Z","iopub.execute_input":"2023-03-26T14:07:40.809735Z","iopub.status.idle":"2023-03-26T14:07:40.834013Z","shell.execute_reply.started":"2023-03-26T14:07:40.809688Z","shell.execute_reply":"2023-03-26T14:07:40.832848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the function below, if I am to explain it in my own words: \n\nWe pad the input video data (out from `load_relevant_data_subset`) to `cfg.INPUT_SIZE #32` frames if number of frames is lower than `cfg.INPUT_SIZE`.","metadata":{}},{"cell_type":"code","source":"class PreprocessLayer(tf.keras.layers.Layer):\n    def __init__(self):\n        super(PreprocessLayer, self).__init__()\n        \n    def pad_edge(self, t, repeats, side):\n        if side == 'LEFT':\n            return tf.concat((tf.repeat(t[:1], repeats=repeats, axis=0), t), axis=0)\n        elif side == 'RIGHT':\n            return tf.concat((t, tf.repeat(t[-1:], repeats=repeats, axis=0)), axis=0)\n    \n    @tf.function(\n        input_signature=(tf.TensorSpec(shape=[None,cfg.N_ROWS,cfg.N_DIMS], dtype=tf.float32),),\n    )\n    def call(self, data0):\n        # Number of Frames in Video\n        N_FRAMES0 = tf.shape(data0)[0]\n        \n        # Keep only non-empty frames in data\n        frames_hands_nansum = tf.experimental.numpy.nanmean(tf.gather(data0, HAND_IDXS0, axis=1), axis=[1,2])\n        non_empty_frames_idxs = tf.where(frames_hands_nansum > 0)\n        non_empty_frames_idxs = tf.squeeze(non_empty_frames_idxs, axis=1)\n        data = tf.gather(data0, non_empty_frames_idxs, axis=0)\n        \n        non_empty_frames_idxs = tf.cast(non_empty_frames_idxs, tf.float32) \n        \n        # Number of non-empty frames\n        N_FRAMES = tf.shape(data)[0]\n        data = tf.gather(data, LANDMARK_IDXS0, axis=1)\n        \n        if N_FRAMES < cfg.INPUT_SIZE:\n            # Video fits in cfg.INPUT_SIZE\n            non_empty_frames_idxs = tf.pad(non_empty_frames_idxs, [[0, cfg.INPUT_SIZE-N_FRAMES]], constant_values=-1)\n            data = tf.pad(data, [[0, cfg.INPUT_SIZE-N_FRAMES], [0,0], [0,0]], constant_values=0)\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            return data, non_empty_frames_idxs\n        else:\n            # Video needs to be downsampled to cfg.INPUT_SIZE\n            if N_FRAMES < cfg.INPUT_SIZE**2:\n                repeats = tf.math.floordiv(cfg.INPUT_SIZE * cfg.INPUT_SIZE, N_FRAMES0)\n                data = tf.repeat(data, repeats=repeats, axis=0)\n                non_empty_frames_idxs = tf.repeat(non_empty_frames_idxs, repeats=repeats, axis=0)\n\n            # Pad To Multiple Of Input Size\n            pool_size = tf.math.floordiv(len(data), cfg.INPUT_SIZE)\n            if tf.math.mod(len(data), cfg.INPUT_SIZE) > 0:\n                pool_size += 1\n            if pool_size == 1:\n                pad_size = (pool_size * cfg.INPUT_SIZE) - len(data)\n            else:\n                pad_size = (pool_size * cfg.INPUT_SIZE) % len(data)\n\n            # Pad Start/End with Start/End value\n            pad_left = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(cfg.INPUT_SIZE, 2)\n            pad_right = tf.math.floordiv(pad_size, 2) + tf.math.floordiv(cfg.INPUT_SIZE, 2)\n            if tf.math.mod(pad_size, 2) > 0:\n                pad_right += 1\n\n            # Pad By Concatenating Left/Right Edge Values\n            data = self.pad_edge(data, pad_left, 'LEFT')\n            data = self.pad_edge(data, pad_right, 'RIGHT')\n\n            # Pad Non Empty Frame Indices\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_left, 'LEFT')\n            non_empty_frames_idxs = self.pad_edge(non_empty_frames_idxs, pad_right, 'RIGHT')\n\n            # Reshape to Mean Pool\n            data = tf.reshape(data, [cfg.INPUT_SIZE, -1, N_COLS, cfg.N_DIMS])\n            non_empty_frames_idxs = tf.reshape(non_empty_frames_idxs, [cfg.INPUT_SIZE, -1])\n\n            # Mean Pool\n            data = tf.experimental.numpy.nanmean(data, axis=1)\n            non_empty_frames_idxs = tf.experimental.numpy.nanmean(non_empty_frames_idxs, axis=1)\n\n            # Fill NaN Values With 0\n            data = tf.where(tf.math.is_nan(data), 0.0, data)\n            \n            return data, non_empty_frames_idxs\n    \npreprocess_layer = PreprocessLayer()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:42.059847Z","iopub.execute_input":"2023-03-26T14:07:42.060275Z","iopub.status.idle":"2023-03-26T14:07:42.104417Z","shell.execute_reply.started":"2023-03-26T14:07:42.060239Z","shell.execute_reply":"2023-03-26T14:07:42.102974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The best way to understand the function above is by calling it. Let's take below data as an example. So the input to the `preprocess_layer` is of shape $(23, 543, 3)$.","metadata":{}},{"cell_type":"markdown","source":"This represents 23 Frames, 543 landmarks (face, hand, pose) & 3 coordinates ($X$, $Y$ & $Z$).","metadata":{}},{"cell_type":"code","source":"sample = load_relevant_data_subset(train.path[0])\nsample.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:47.446137Z","iopub.execute_input":"2023-03-26T14:07:47.446514Z","iopub.status.idle":"2023-03-26T14:07:47.530553Z","shell.execute_reply.started":"2023-03-26T14:07:47.446463Z","shell.execute_reply":"2023-03-26T14:07:47.529553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data, non_empty_frames_idxs = preprocess_layer(sample)\ndata.shape, non_empty_frames_idxs","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:07:47.968960Z","iopub.execute_input":"2023-03-26T14:07:47.971449Z","iopub.status.idle":"2023-03-26T14:07:51.322304Z","shell.execute_reply.started":"2023-03-26T14:07:47.971413Z","shell.execute_reply":"2023-03-26T14:07:51.321153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# free up RAM, delete variables as we go\ndel data; del non_empty_frames_idxs","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:08:18.378880Z","iopub.execute_input":"2023-03-26T14:08:18.379567Z","iopub.status.idle":"2023-03-26T14:08:18.386010Z","shell.execute_reply.started":"2023-03-26T14:08:18.379472Z","shell.execute_reply":"2023-03-26T14:08:18.384881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As can be seen above the output is of shape $(32, 92, 3)$, can you guess why? Take a minute to read the function above and try to guess why. It will help in your understanding of the code. :) ","metadata":{}},{"cell_type":"markdown","source":"In the output 32 represents the interpolated frame size, 92 key landmarks that we keep instead of 543 and 3 coordinates - $X$, $Y$ and $Z$.","metadata":{}},{"cell_type":"markdown","source":"What about when `N_FRAMES < cfg.INPUT_SIZE` is False? I leave that as an exercise for you. :) ","metadata":{}},{"cell_type":"markdown","source":"## Create Dataset","metadata":{}},{"cell_type":"code","source":"LIPS_START = 0\nLEFT_HAND_START = LIPS_IDXS.size\nRIGHT_HAND_START = LEFT_HAND_START + LEFT_HAND_IDXS.size\nPOSE_START = RIGHT_HAND_START + RIGHT_HAND_IDXS.size","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:08:27.001223Z","iopub.execute_input":"2023-03-26T14:08:27.001622Z","iopub.status.idle":"2023-03-26T14:08:27.006762Z","shell.execute_reply.started":"2023-03-26T14:08:27.001587Z","shell.execute_reply":"2023-03-26T14:08:27.005560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_data(file_path):\n    data = load_relevant_data_subset(file_path)\n    data = preprocess_layer(data)\n    return data","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:08:27.416179Z","iopub.execute_input":"2023-03-26T14:08:27.417032Z","iopub.status.idle":"2023-03-26T14:08:27.423663Z","shell.execute_reply.started":"2023-03-26T14:08:27.416983Z","shell.execute_reply":"2023-03-26T14:08:27.422562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_x_y():\n    # Create arrays to save data\n    X = np.zeros([N_SAMPLES, INPUT_SIZE, N_COLS, N_DIMS], dtype=np.float32)\n    y = np.zeros([N_SAMPLES], dtype=np.int32)\n    NON_EMPTY_FRAME_IDXS = np.full([N_SAMPLES, INPUT_SIZE], -1, dtype=np.float32)\n\n    for row_idx, (file_path, sign_ord) in enumerate(tqdm(train[['file_path', 'sign_ord']].values)):\n        if row_idx % 5000 == 0:\n            print(f'Generated {row_idx}/{N_SAMPLES}')\n\n        data, non_empty_frame_idxs = get_data(file_path)\n        X[row_idx] = data\n        y[row_idx] = sign_ord\n        NON_EMPTY_FRAME_IDXS[row_idx] = non_empty_frame_idxs\n        if np.isnan(data).sum() > 0: return data\n\n    # Save X/y\n    np.save('X.npy', X)\n    np.save('y.npy', y)\n    np.save('NON_EMPTY_FRAME_IDXS.npy', NON_EMPTY_FRAME_IDXS)\n    \n    return X, y, NON_EMPTY_FRAME_IDXS","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:08:28.913408Z","iopub.execute_input":"2023-03-26T14:08:28.920734Z","iopub.status.idle":"2023-03-26T14:08:28.934541Z","shell.execute_reply.started":"2023-03-26T14:08:28.920686Z","shell.execute_reply":"2023-03-26T14:08:28.933439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if cfg.PREPROCESS_DATA:\n    X, y, NON_EMPTY_FRAME_IDXS = get_x_y()\nelse:\n    X = np.load('/kaggle/input/gislr-dataset-public/X.npy')\n    y = np.load('/kaggle/input/gislr-dataset-public/y.npy')\n    NON_EMPTY_FRAME_IDXS = np.load('/kaggle/input/gislr-dataset-public/NON_EMPTY_FRAME_IDXS.npy')","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:08:42.659311Z","iopub.execute_input":"2023-03-26T14:08:42.660332Z","iopub.status.idle":"2023-03-26T14:09:07.411817Z","shell.execute_reply.started":"2023-03-26T14:08:42.660289Z","shell.execute_reply":"2023-03-26T14:09:07.410692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape, y.shape, NON_EMPTY_FRAME_IDXS.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:07.417027Z","iopub.execute_input":"2023-03-26T14:09:07.417714Z","iopub.status.idle":"2023-03-26T14:09:07.431099Z","shell.execute_reply.started":"2023-03-26T14:09:07.417667Z","shell.execute_reply":"2023-03-26T14:09:07.429969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Statistics - Lips","metadata":{}},{"cell_type":"code","source":"# LIPS\nLIPS_MEAN_X  = np.zeros([LIPS_IDXS.size], dtype=np.float32)\nLIPS_MEAN_Y  = np.zeros([LIPS_IDXS.size], dtype=np.float32)\nLIPS_STD_X   = np.zeros([LIPS_IDXS.size], dtype=np.float32)\nLIPS_STD_Y   = np.zeros([LIPS_IDXS.size], dtype=np.float32)\n\nfor col, ll in enumerate(tqdm( np.transpose(X[:,:,LIPS_IDXS], [2,3,0,1]).reshape([LIPS_IDXS.size, cfg.N_DIMS, -1]) )):\n    for dim, l in enumerate(ll):\n        v = l[np.nonzero(l)]\n        if dim == 0: # X\n            LIPS_MEAN_X[col] = v.mean()\n            LIPS_STD_X[col] = v.std()\n        if dim == 1: # Y\n            LIPS_MEAN_Y[col] = v.mean()\n            LIPS_STD_Y[col] = v.std()\n        \nLIPS_MEAN = np.array([LIPS_MEAN_X, LIPS_MEAN_Y]).T\nLIPS_STD = np.array([LIPS_STD_X, LIPS_STD_Y]).T","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:07.433407Z","iopub.execute_input":"2023-03-26T14:09:07.435210Z","iopub.status.idle":"2023-03-26T14:09:16.153623Z","shell.execute_reply.started":"2023-03-26T14:09:07.435175Z","shell.execute_reply":"2023-03-26T14:09:16.152551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Statistics - Hands","metadata":{}},{"cell_type":"code","source":"# LEFT HAND\nLEFT_HANDS_MEAN_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\nLEFT_HANDS_MEAN_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\nLEFT_HANDS_STD_X = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\nLEFT_HANDS_STD_Y = np.zeros([LEFT_HAND_IDXS.size], dtype=np.float32)\n# RIGHT HAND\nRIGHT_HANDS_MEAN_X = np.zeros([RIGHT_HAND_IDXS.size], dtype=np.float32)\nRIGHT_HANDS_MEAN_Y = np.zeros([RIGHT_HAND_IDXS.size], dtype=np.float32)\nRIGHT_HANDS_STD_X = np.zeros([RIGHT_HAND_IDXS.size], dtype=np.float32)\nRIGHT_HANDS_STD_Y = np.zeros([RIGHT_HAND_IDXS.size], dtype=np.float32)\n\nfor col, ll in enumerate(tqdm( np.transpose(X[:,:,HAND_IDXS], [2,3,0,1]).reshape([HAND_IDXS.size, cfg.N_DIMS, -1]) )):\n    for dim, l in enumerate(ll):\n        v = l[np.nonzero(l)]\n        if dim == 0: # X\n            if col < RIGHT_HAND_IDXS.size: # LEFT HAND\n                LEFT_HANDS_MEAN_X[col] = v.mean()\n                LEFT_HANDS_STD_X[col] = v.std()\n            else:\n                RIGHT_HANDS_MEAN_X[col - LEFT_HAND_IDXS.size] = v.mean()\n                RIGHT_HANDS_STD_X[col - LEFT_HAND_IDXS.size] = v.std()\n        if dim == 1: # Y\n            if col < RIGHT_HAND_IDXS.size: # LEFT HAND\n                LEFT_HANDS_MEAN_Y[col] = v.mean()\n                LEFT_HANDS_STD_Y[col] = v.std()\n            else: # RIGHT HAND\n                RIGHT_HANDS_MEAN_Y[col - LEFT_HAND_IDXS.size] = v.mean()\n                RIGHT_HANDS_STD_Y[col - LEFT_HAND_IDXS.size] = v.std()\n        \nLEFT_HANDS_MEAN = np.array([LEFT_HANDS_MEAN_X, LEFT_HANDS_MEAN_Y]).T\nLEFT_HANDS_STD = np.array([LEFT_HANDS_STD_X, LEFT_HANDS_STD_Y]).T\nRIGHT_HANDS_MEAN = np.array([RIGHT_HANDS_MEAN_X, RIGHT_HANDS_MEAN_Y]).T\nRIGHT_HANDS_STD = np.array([RIGHT_HANDS_STD_X, RIGHT_HANDS_STD_Y]).T","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:16.156544Z","iopub.execute_input":"2023-03-26T14:09:16.157189Z","iopub.status.idle":"2023-03-26T14:09:24.947554Z","shell.execute_reply.started":"2023-03-26T14:09:16.157149Z","shell.execute_reply":"2023-03-26T14:09:24.946558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Statistics - Pose","metadata":{}},{"cell_type":"code","source":"# POSE\nPOSE_MEAN_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\nPOSE_MEAN_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\nPOSE_STD_X = np.zeros([POSE_IDXS.size], dtype=np.float32)\nPOSE_STD_Y = np.zeros([POSE_IDXS.size], dtype=np.float32)\n\nfor col, ll in enumerate(tqdm( np.transpose(X[:,:,POSE_IDXS], [2,3,0,1]).reshape([POSE_IDXS.size, cfg.N_DIMS, -1]) )):\n    for dim, l in enumerate(ll):\n        v = l[np.nonzero(l)]\n        if dim == 0: # X\n            POSE_MEAN_X[col] = v.mean()\n            POSE_STD_X[col] = v.std()\n        if dim == 1: # Y\n            POSE_MEAN_Y[col] = v.mean()\n            POSE_STD_Y[col] = v.std()\n        \nPOSE_MEAN = np.array([POSE_MEAN_X, POSE_MEAN_Y]).T\nPOSE_STD = np.array([POSE_STD_X, POSE_STD_Y]).T","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:24.949075Z","iopub.execute_input":"2023-03-26T14:09:24.949550Z","iopub.status.idle":"2023-03-26T14:09:27.104611Z","shell.execute_reply.started":"2023-03-26T14:09:24.949510Z","shell.execute_reply":"2023-03-26T14:09:27.103563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Samples","metadata":{}},{"cell_type":"code","source":"# Custom sampler to get a batch containing N times all signs\ndef get_train_batch_all_signs(X, y, NON_EMPTY_FRAME_IDXS, n=cfg.BATCH_ALL_SIGNS_N):\n    # Arrays to store batch in\n    X_batch = np.zeros([cfg.NUM_CLASSES*n, cfg.INPUT_SIZE, N_COLS, cfg.N_DIMS], dtype=np.float32)\n    y_batch = np.arange(0, cfg.NUM_CLASSES, step=1/n, dtype=np.float32).astype(np.int64)\n    non_empty_frame_idxs_batch = np.zeros([cfg.NUM_CLASSES*n, cfg.INPUT_SIZE], dtype=np.float32)\n    \n    # Dictionary mapping ordinally encoded sign to corresponding sample indices\n    CLASS2IDXS = {}\n    for i in range(cfg.NUM_CLASSES):\n        CLASS2IDXS[i] = np.argwhere(y == i).squeeze().astype(np.int32)\n            \n    while True:\n        # Fill batch arrays\n        for i in range(cfg.NUM_CLASSES):\n            idxs = np.random.choice(CLASS2IDXS[i], n)\n            X_batch[i*n:(i+1)*n] = X[idxs]\n            non_empty_frame_idxs_batch[i*n:(i+1)*n] = NON_EMPTY_FRAME_IDXS[idxs]\n        \n        yield { 'frames': X_batch, 'non_empty_frame_idxs': non_empty_frame_idxs_batch }, y_batch","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:27.106215Z","iopub.execute_input":"2023-03-26T14:09:27.106884Z","iopub.status.idle":"2023-03-26T14:09:27.117112Z","shell.execute_reply.started":"2023-03-26T14:09:27.106845Z","shell.execute_reply":"2023-03-26T14:09:27.115828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dummy_dataset = get_train_batch_all_signs(X, y, NON_EMPTY_FRAME_IDXS)\nX_batch, y_batch = next(dummy_dataset)\nX_batch.keys(), X_batch['frames'].shape","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:27.118904Z","iopub.execute_input":"2023-03-26T14:09:27.119322Z","iopub.status.idle":"2023-03-26T14:09:27.182756Z","shell.execute_reply.started":"2023-03-26T14:09:27.119280Z","shell.execute_reply":"2023-03-26T14:09:27.181569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Config","metadata":{}},{"cell_type":"code","source":"# Epsilon value for layer normalisation\nLAYER_NORM_EPS = 1e-6\n\n# Dense layer units for landmarks\nLIPS_UNITS = 384\nHANDS_UNITS = 384\nPOSE_UNITS = 384\n# final embedding and transformer embedding size\nUNITS = 512\n\n# Transformer\nNUM_BLOCKS = 2\nMLP_RATIO = 2\n\n# Dropout\nEMBEDDING_DROPOUT = 0.00\nMLP_DROPOUT_RATIO = 0.30\nCLASSIFIER_DROPOUT_RATIO = 0.10\n\n# Initiailizers\nINIT_HE_UNIFORM = tf.keras.initializers.he_uniform\nINIT_GLOROT_UNIFORM = tf.keras.initializers.glorot_uniform\nINIT_ZEROS = tf.keras.initializers.constant(0.0)\n# Activations\nGELU = tf.keras.activations.gelu","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:27.184257Z","iopub.execute_input":"2023-03-26T14:09:27.184713Z","iopub.status.idle":"2023-03-26T14:09:27.192103Z","shell.execute_reply.started":"2023-03-26T14:09:27.184676Z","shell.execute_reply":"2023-03-26T14:09:27.190694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Transformer\n\nNeed to implement transformer from scratch as TFLite does not support the native TF implementation of MultiHeadAttention.","metadata":{}},{"cell_type":"code","source":"# based on: https://stackoverflow.com/questions/67342988/verifying-the-implementation-of-multihead-attention-in-transformer\n# replaced softmax with softmax layer to support masked softmax\ndef scaled_dot_product(q,k,v, softmax, attention_mask):\n    #calculates Q . K(transpose)\n    qkt = tf.matmul(q,k,transpose_b=True)\n    #caculates scaling factor\n    dk = tf.math.sqrt(tf.cast(q.shape[-1],dtype=tf.float32))\n    scaled_qkt = qkt/dk\n    softmax = softmax(scaled_qkt, mask=attention_mask)\n    \n    z = tf.matmul(softmax,v)\n    #shape: (m,Tx,depth), same shape as q,k,v\n    return z\n\nclass MultiHeadAttention(tf.keras.layers.Layer):\n    def __init__(self,d_model,num_of_heads):\n        super(MultiHeadAttention,self).__init__()\n        self.d_model = d_model\n        self.num_of_heads = num_of_heads\n        self.depth = d_model//num_of_heads\n        self.wq = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wk = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wv = [tf.keras.layers.Dense(self.depth) for i in range(num_of_heads)]\n        self.wo = tf.keras.layers.Dense(d_model)\n        self.softmax = tf.keras.layers.Softmax()\n        \n    def call(self,x, attention_mask):\n        \n        multi_attn = []\n        for i in range(self.num_of_heads):\n            Q = self.wq[i](x)\n            K = self.wk[i](x)\n            V = self.wv[i](x)\n            multi_attn.append(scaled_dot_product(Q,K,V, self.softmax, attention_mask))\n            \n        multi_head = tf.concat(multi_attn,axis=-1)\n        multi_head_attention = self.wo(multi_head)\n        return multi_head_attention","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:27.193758Z","iopub.execute_input":"2023-03-26T14:09:27.194257Z","iopub.status.idle":"2023-03-26T14:09:27.208433Z","shell.execute_reply.started":"2023-03-26T14:09:27.194220Z","shell.execute_reply":"2023-03-26T14:09:27.207554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Transformer(tf.keras.Model):\n    def __init__(self, num_blocks):\n        super(Transformer, self).__init__(name='transformer')\n        self.num_blocks = num_blocks\n    \n    def build(self, input_shape):\n        self.ln_1s = []\n        self.mhas = []\n        self.ln_2s = []\n        self.mlps = []\n        # Make Transformer Blocks\n        for i in range(self.num_blocks):\n            # First Layer Normalisation\n            self.ln_1s.append(tf.keras.layers.LayerNormalization(epsilon=LAYER_NORM_EPS))\n            # Multi Head Attention\n            self.mhas.append(MultiHeadAttention(UNITS, 4))\n            # Second Layer Normalisation\n            self.ln_2s.append(tf.keras.layers.LayerNormalization(epsilon=LAYER_NORM_EPS))\n            # Multi Layer Perception\n            self.mlps.append(tf.keras.Sequential([\n                tf.keras.layers.Dense(UNITS * MLP_RATIO, activation=GELU, kernel_initializer=INIT_GLOROT_UNIFORM),\n                tf.keras.layers.Dropout(MLP_DROPOUT_RATIO),\n                tf.keras.layers.Dense(UNITS, kernel_initializer=INIT_HE_UNIFORM),\n            ]))\n        \n    def call(self, x, attention_mask):\n        # Iterate input over transformer blocks\n        for ln_1, mha, ln_2, mlp in zip(self.ln_1s, self.mhas, self.ln_2s, self.mlps):\n            x1 = ln_1(x)\n            attention_output = mha(x1, attention_mask)\n            x2 = x1 + attention_output\n            x3 = ln_2(x2)\n            x3 = mlp(x3)\n            x = x3 + x2\n    \n        return x","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:27.212479Z","iopub.execute_input":"2023-03-26T14:09:27.213062Z","iopub.status.idle":"2023-03-26T14:09:27.224314Z","shell.execute_reply.started":"2023-03-26T14:09:27.213023Z","shell.execute_reply":"2023-03-26T14:09:27.223260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Landmark Embedding","metadata":{}},{"cell_type":"code","source":"class LandmarkEmbedding(tf.keras.Model):\n    def __init__(self, units, name):\n        super(LandmarkEmbedding, self).__init__(name=f'{name}_embedding')\n        self.units = units\n        \n    def build(self, input_shape):\n        # Embedding for missing landmark in frame, initizlied with zeros\n        self.empty_embedding = self.add_weight(\n            name=f'{self.name}_empty_embedding',\n            shape=[self.units],\n            initializer=INIT_ZEROS,\n        )\n        # Embedding\n        self.dense = tf.keras.Sequential([\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM, activation=GELU),\n            tf.keras.layers.Dense(self.units, name=f'{self.name}_dense_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name=f'{self.name}_dense')\n\n    def call(self, x):\n        return tf.where(\n                # Checks whether landmark is missing in frame\n                tf.reduce_sum(x, axis=2, keepdims=True) == 0,\n                # If so, the empty embedding is used\n                self.empty_embedding,\n                # Otherwise the landmark data is embedded\n                self.dense(x),\n            )","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:27.225983Z","iopub.execute_input":"2023-03-26T14:09:27.226386Z","iopub.status.idle":"2023-03-26T14:09:27.238606Z","shell.execute_reply.started":"2023-03-26T14:09:27.226350Z","shell.execute_reply":"2023-03-26T14:09:27.237782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Embedding","metadata":{}},{"cell_type":"code","source":"class CustomEmbedding(tf.keras.Model):\n    def __init__(self):\n        super(CustomEmbedding, self).__init__()\n        \n    def get_diffs(self, l):\n        S = l.shape[2]\n        other = tf.expand_dims(l, 3)\n        other = tf.repeat(other, S, axis=3)\n        other = tf.transpose(other, [0,1,3,2])\n        diffs = tf.expand_dims(l, 3) - other\n        diffs = tf.reshape(diffs, [-1, cfg.INPUT_SIZE, S*S])\n        return diffs\n\n    def build(self, input_shape):\n        # Positional Embedding, initialized with zeros\n        self.positional_embedding = tf.keras.layers.Embedding(cfg.INPUT_SIZE+1, UNITS, embeddings_initializer=INIT_ZEROS)\n        # Embedding layer for Landmarks\n        self.lips_embedding = LandmarkEmbedding(LIPS_UNITS, 'lips')\n        self.left_hand_embedding = LandmarkEmbedding(HANDS_UNITS, 'left_hand')\n        self.right_hand_embedding = LandmarkEmbedding(HANDS_UNITS, 'right_hand')\n        self.pose_embedding = LandmarkEmbedding(POSE_UNITS, 'pose')\n        # Landmark Weights\n        self.landmark_weights = tf.Variable(tf.zeros([4], dtype=tf.float32), name='landmark_weights')\n        # Fully Connected Layers for combined landmarks\n        self.fc = tf.keras.Sequential([\n            tf.keras.layers.Dense(UNITS, name='fully_connected_1', use_bias=False, kernel_initializer=INIT_GLOROT_UNIFORM, activation=GELU),\n            tf.keras.layers.Dense(UNITS, name='fully_connected_2', use_bias=False, kernel_initializer=INIT_HE_UNIFORM),\n        ], name='fc')\n\n\n    def call(self, lips0, left_hand0, right_hand0, pose0, non_empty_frame_idxs, training=False):\n        # Lips\n        lips_embedding = self.lips_embedding(lips0)\n        # Left Hand\n        left_hand_embedding = self.left_hand_embedding(left_hand0)\n        # Right Hand\n        right_hand_embedding = self.right_hand_embedding(right_hand0)\n        # Pose\n        pose_embedding = self.pose_embedding(pose0)\n        # Merge Embeddings of all landmarks with mean pooling\n        x = tf.stack((lips_embedding, left_hand_embedding, right_hand_embedding, pose_embedding), axis=3)\n        # Merge Landmarks with trainable attention weights\n        x = x * tf.nn.softmax(self.landmark_weights)\n        x = tf.reduce_sum(x, axis=3)\n        # Fully Connected Layers\n        x = self.fc(x)\n        # Add Positional Embedding\n        normalised_non_empty_frame_idxs = tf.where(\n            tf.math.equal(non_empty_frame_idxs, -1.0),\n            cfg.INPUT_SIZE,\n            tf.cast(\n                non_empty_frame_idxs / tf.reduce_max(non_empty_frame_idxs, axis=1, keepdims=True) * cfg.INPUT_SIZE,\n                tf.int32,\n            ),\n        )\n        x = x + self.positional_embedding(normalised_non_empty_frame_idxs)\n        \n        return x","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:27.240144Z","iopub.execute_input":"2023-03-26T14:09:27.240928Z","iopub.status.idle":"2023-03-26T14:09:27.256069Z","shell.execute_reply.started":"2023-03-26T14:09:27.240891Z","shell.execute_reply":"2023-03-26T14:09:27.255028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_lr_metric(optimizer):\n    def lr(y_true, y_pred):\n        return optimizer.lr\n    return lr","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:27.257444Z","iopub.execute_input":"2023-03-26T14:09:27.258025Z","iopub.status.idle":"2023-03-26T14:09:27.269966Z","shell.execute_reply.started":"2023-03-26T14:09:27.257986Z","shell.execute_reply":"2023-03-26T14:09:27.269055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    # Inputs\n    frames = tf.keras.layers.Input([cfg.INPUT_SIZE, N_COLS, cfg.N_DIMS], dtype=tf.float32, name='frames')\n    non_empty_frame_idxs = tf.keras.layers.Input([cfg.INPUT_SIZE], dtype=tf.float32, name='non_empty_frame_idxs')\n    # Padding Mask\n    mask = tf.cast(tf.math.not_equal(non_empty_frame_idxs, -1), tf.float32)\n    mask = tf.expand_dims(mask, axis=2)\n    \n    x = frames\n    x = tf.slice(x, [0,0,0,0], [-1,cfg.INPUT_SIZE, N_COLS, 2])\n    # LIPS\n    lips = tf.slice(x, [0,0,LIPS_START,0], [-1,cfg.INPUT_SIZE, 40, 2])\n    lips = tf.where(\n            tf.math.equal(lips, 0.0),\n            0.0,\n            (lips - LIPS_MEAN) / LIPS_STD,\n        )\n    lips = tf.reshape(lips, [-1, cfg.INPUT_SIZE, 40*2])\n    # LEFT HAND\n    left_hand = tf.slice(x, [0,0,40,0], [-1,cfg.INPUT_SIZE, 21, 2])\n    left_hand = tf.where(\n            tf.math.equal(left_hand, 0.0),\n            0.0,\n            (left_hand - LEFT_HANDS_MEAN) / LEFT_HANDS_STD,\n        )\n    left_hand = tf.reshape(left_hand, [-1, cfg.INPUT_SIZE, 21*2])\n    # RIGHT HAND\n    right_hand = tf.slice(x, [0,0,61,0], [-1,cfg.INPUT_SIZE, 21, 2])\n    right_hand = tf.where(\n            tf.math.equal(right_hand, 0.0),\n            0.0,\n            (right_hand - RIGHT_HANDS_MEAN) / RIGHT_HANDS_STD,\n        )\n    right_hand = tf.reshape(right_hand, [-1, cfg.INPUT_SIZE, 21*2])\n    # POSE\n    pose = tf.slice(x, [0,0,82,0], [-1,cfg.INPUT_SIZE, 10, 2])\n    pose = tf.where(\n            tf.math.equal(pose, 0.0),\n            0.0,\n            (pose - POSE_MEAN) / POSE_STD,\n        )\n    pose = tf.reshape(pose, [-1, cfg.INPUT_SIZE, 10*2])\n    x = lips, left_hand, right_hand, pose\n    x = CustomEmbedding()(lips, left_hand, right_hand, pose, non_empty_frame_idxs)\n    # Encoder Transformer Blocks\n    x = Transformer(NUM_BLOCKS)(x, mask)\n    # Pooling\n    x = tf.reduce_sum(x * mask, axis=1) / tf.reduce_sum(mask, axis=1)\n    # Classification Layer\n    x = tf.keras.layers.Dense(cfg.NUM_CLASSES, activation=tf.keras.activations.softmax, kernel_initializer=INIT_GLOROT_UNIFORM)(x)\n    outputs = x\n    \n    # Create Tensorflow Model\n    model = tf.keras.models.Model(inputs=[frames, non_empty_frame_idxs], outputs=outputs)\n    \n    # Simple Categorical Crossentropy Loss\n    loss = tf.keras.losses.SparseCategoricalCrossentropy()\n    \n    # Adam Optimizer with weight decay\n    optimizer = tfa.optimizers.AdamW(learning_rate=1e-3, weight_decay=1e-5, clipnorm=1.0)\n    \n    lr_metric = get_lr_metric(optimizer)\n    metrics = [\"acc\",lr_metric]\n    model.compile(loss=loss, optimizer=optimizer, metrics=metrics)\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:27.271587Z","iopub.execute_input":"2023-03-26T14:09:27.271947Z","iopub.status.idle":"2023-03-26T14:09:27.402432Z","shell.execute_reply.started":"2023-03-26T14:09:27.271912Z","shell.execute_reply":"2023-03-26T14:09:27.401030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.backend.clear_session()\n\nmodel_one = get_model()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:27.404001Z","iopub.execute_input":"2023-03-26T14:09:27.405126Z","iopub.status.idle":"2023-03-26T14:09:29.831967Z","shell.execute_reply.started":"2023-03-26T14:09:27.405087Z","shell.execute_reply":"2023-03-26T14:09:29.830913Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_one.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:30.812982Z","iopub.execute_input":"2023-03-26T14:09:30.813835Z","iopub.status.idle":"2023-03-26T14:09:31.070595Z","shell.execute_reply.started":"2023-03-26T14:09:30.813787Z","shell.execute_reply":"2023-03-26T14:09:31.069748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Learning Rate Scheduler","metadata":{}},{"cell_type":"code","source":"def lrfn(current_step, num_warmup_steps, lr_max, num_cycles=0.50, num_training_steps=cfg.N_EPOCHS):\n    \n    if current_step < num_warmup_steps:\n        if WARMUP_METHOD == 'log':\n            return lr_max * 0.10 ** (num_warmup_steps - current_step)\n        else:\n            return lr_max * 2 ** -(num_warmup_steps - current_step)\n    else:\n        progress = float(current_step - num_warmup_steps) / float(max(1, num_training_steps - num_warmup_steps))\n\n        return max(0.0, 0.5 * (1.0 + math.cos(math.pi * float(num_cycles) * 2.0 * progress))) * lr_max","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:39.704284Z","iopub.execute_input":"2023-03-26T14:09:39.705001Z","iopub.status.idle":"2023-03-26T14:09:39.712931Z","shell.execute_reply.started":"2023-03-26T14:09:39.704962Z","shell.execute_reply":"2023-03-26T14:09:39.710715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_lr_schedule(lr_schedule, epochs):\n    fig = plt.figure(figsize=(20, 10))\n    plt.plot([None] + lr_schedule + [None])\n    # X Labels\n    x = np.arange(1, epochs + 1)\n    x_axis_labels = [i if epochs <= 40 or i % 5 == 0 or i == 1 else None for i in range(1, epochs + 1)]\n    plt.xlim([1, epochs])\n    plt.xticks(x, x_axis_labels) # set tick step to 1 and let x axis start at 1\n    \n    # Increase y-limit for better readability\n    plt.ylim([0, max(lr_schedule) * 1.1])\n    \n    # Title\n    schedule_info = f'start: {lr_schedule[0]:.1E}, max: {max(lr_schedule):.1E}, final: {lr_schedule[-1]:.1E}'\n    plt.title(f'Step Learning Rate Schedule, {schedule_info}', size=18, pad=12)\n    \n    # Plot Learning Rates\n    for x, val in enumerate(lr_schedule):\n        if epochs <= 40 or x % 5 == 0 or x is epochs - 1:\n            if x < len(lr_schedule) - 1:\n                if lr_schedule[x - 1] < val:\n                    ha = 'right'\n                else:\n                    ha = 'left'\n            elif x == 0:\n                ha = 'right'\n            else:\n                ha = 'left'\n            plt.plot(x + 1, val, 'o', color='black');\n            offset_y = (max(lr_schedule) - min(lr_schedule)) * 0.02\n            plt.annotate(f'{val:.1E}', xy=(x + 1, val + offset_y), size=12, ha=ha)\n    \n    plt.xlabel('Epoch', size=16, labelpad=5)\n    plt.ylabel('Learning Rate', size=16, labelpad=5)\n    plt.grid()\n    plt.show()\n\n# Learning rate for encoder\nLR_SCHEDULE = [lrfn(step, num_warmup_steps=cfg.N_WARMUP_EPOCHS, lr_max=cfg.LR_MAX, num_cycles=0.50) for step in range(cfg.N_EPOCHS)]\n# Plot Learning Rate Schedule\nplot_lr_schedule(LR_SCHEDULE, epochs=cfg.N_EPOCHS)\n# Learning Rate Callback\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lambda step: LR_SCHEDULE[step], verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:56.044636Z","iopub.execute_input":"2023-03-26T14:09:56.045745Z","iopub.status.idle":"2023-03-26T14:09:56.562574Z","shell.execute_reply.started":"2023-03-26T14:09:56.045696Z","shell.execute_reply":"2023-03-26T14:09:56.561565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Weight Decay Callback","metadata":{}},{"cell_type":"code","source":"# Custom callback to update weight decay with learning rate\nclass WeightDecayCallback(tf.keras.callbacks.Callback):\n    def __init__(self, wd_ratio=cfg.WD_RATIO):\n        self.step_counter = 0\n        self.wd_ratio = wd_ratio\n    \n    def on_epoch_begin(self, epoch, logs=None):\n        model_one.optimizer.weight_decay = model_one.optimizer.learning_rate * self.wd_ratio\n        print(f'learning rate: {model_one.optimizer.learning_rate.numpy():.2e}, weight decay: {model_one.optimizer.weight_decay.numpy():.2e}')","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:09:58.098378Z","iopub.execute_input":"2023-03-26T14:09:58.099078Z","iopub.status.idle":"2023-03-26T14:09:58.105703Z","shell.execute_reply.started":"2023-03-26T14:09:58.099041Z","shell.execute_reply":"2023-03-26T14:09:58.104189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Performance Benchmark","metadata":{}},{"cell_type":"code","source":"%%timeit -n 100\nif cfg.TRAIN_MODEL:\n    # Verify model_one prediction is <<<100ms\n    model_one.predict_on_batch({ 'frames': X[:1], 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS[:1] })","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:10:00.668400Z","iopub.execute_input":"2023-03-26T14:10:00.669443Z","iopub.status.idle":"2023-03-26T14:10:00.677320Z","shell.execute_reply.started":"2023-03-26T14:10:00.669402Z","shell.execute_reply":"2023-03-26T14:10:00.676019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"X_train = X[train_idxs]\nX_val = X[val_idxs]\nNON_EMPTY_FRAME_IDXS_TRAIN = NON_EMPTY_FRAME_IDXS[train_idxs]\nNON_EMPTY_FRAME_IDXS_VAL = NON_EMPTY_FRAME_IDXS[val_idxs]\ny_train = y[train_idxs]\ny_val = y[val_idxs]","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:10:04.226019Z","iopub.execute_input":"2023-03-26T14:10:04.226416Z","iopub.status.idle":"2023-03-26T14:10:06.116348Z","shell.execute_reply.started":"2023-03-26T14:10:04.226380Z","shell.execute_reply":"2023-03-26T14:10:06.115209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if cfg.TRAIN_MODEL:\n    run = wandb.init(project=\"kaggle-asl-signs\", config=cfg, tags=['transformer', 'final-model'])","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:10:06.118459Z","iopub.execute_input":"2023-03-26T14:10:06.118842Z","iopub.status.idle":"2023-03-26T14:10:06.126812Z","shell.execute_reply.started":"2023-03-26T14:10:06.118810Z","shell.execute_reply":"2023-03-26T14:10:06.123090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if cfg.TRAIN_MODEL:\n    tf.keras.backend.clear_session()\n    callbacks=[\n            lr_callback,\n            WeightDecayCallback(),\n            wandb.keras.WandbCallback()\n    ]\n    model.fit(\n        x=get_train_batch_all_signs(X, y, NON_EMPTY_FRAME_IDXS),\n        steps_per_epoch=len(X) // (NUM_CLASSES * BATCH_ALL_SIGNS_N),\n        epochs=cfg.N_EPOCHS,\n        batch_size=BATCH_SIZE,\n        callbacks=callbacks,\n        verbose = 2,) ","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-03-26T14:10:06.128151Z","iopub.execute_input":"2023-03-26T14:10:06.129568Z","iopub.status.idle":"2023-03-26T14:10:06.136653Z","shell.execute_reply.started":"2023-03-26T14:10:06.129530Z","shell.execute_reply":"2023-03-26T14:10:06.135520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save model artifacts\nif cfg.TRAIN_MODEL:\n    model.save('./final_model_one')\n    model.save_weights('./final_model_one_weights')","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:10:06.343120Z","iopub.execute_input":"2023-03-26T14:10:06.344177Z","iopub.status.idle":"2023-03-26T14:10:06.349410Z","shell.execute_reply.started":"2023-03-26T14:10:06.344123Z","shell.execute_reply":"2023-03-26T14:10:06.348072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# upload artifact to W&B\nif cfg.TRAIN_MODEL:\n    artifact = wandb.Artifact('final_model_one', type='model')\n    artifact.add_file('./final_model_one_weights.data-00000-of-00001')\n    artifact.add_file('./final_model_one_weights.index')\n    run.log_artifact(artifact)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:10:08.144479Z","iopub.execute_input":"2023-03-26T14:10:08.145480Z","iopub.status.idle":"2023-03-26T14:10:08.151355Z","shell.execute_reply.started":"2023-03-26T14:10:08.145441Z","shell.execute_reply":"2023-03-26T14:10:08.150154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# download model\ntry:\n    artifact = run.use_artifact('amanarora/asl-sings/final_model_one:v2', type='model')\nexcept NameError:\n    run = wandb.init(project=\"kaggle-asl-signs\")\n    artifact = run.use_artifact('amanarora/asl-sings/final_model_one:v2', type='model')\nartifact_dir = artifact.download()\n!ls {artifact_dir}","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:10:13.691267Z","iopub.execute_input":"2023-03-26T14:10:13.692011Z","iopub.status.idle":"2023-03-26T14:10:53.559734Z","shell.execute_reply.started":"2023-03-26T14:10:13.691971Z","shell.execute_reply":"2023-03-26T14:10:53.558378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load model weights and do sanity c\nmodel_one.load_weights(f\"{artifact_dir}/final_model_one_weights\")\ny_val_pred = model_one.predict({ 'frames': X_val, 'non_empty_frame_idxs': NON_EMPTY_FRAME_IDXS_VAL }, verbose=2).argmax(axis=1)\nnp.mean(y_val == y_val_pred)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:10:53.565708Z","iopub.execute_input":"2023-03-26T14:10:53.568250Z","iopub.status.idle":"2023-03-26T14:11:06.142042Z","shell.execute_reply.started":"2023-03-26T14:10:53.568202Z","shell.execute_reply":"2023-03-26T14:11:06.140558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# delete variables as we go to free up RAM\ndel X; del y; del X_train; del X_val; del y_train; del y_val","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:11:21.674555Z","iopub.execute_input":"2023-03-26T14:11:21.675169Z","iopub.status.idle":"2023-03-26T14:11:23.410795Z","shell.execute_reply.started":"2023-03-26T14:11:21.675131Z","shell.execute_reply":"2023-03-26T14:11:23.409403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Landmark Attention Weights","metadata":{}},{"cell_type":"markdown","source":"By checking attention weights as from the original notebook, we can check that the left hand and right hand signals are the most important. This makes sense right? ","metadata":{}},{"cell_type":"code","source":"# Landmark Weights\nweights = scipy.special.softmax(model_one.get_layer('custom_embedding').weights[15])\nlandmarks = ['lips_embedding', 'left_hand_embedding', 'right_hand_embedding', 'pose_embedding']\n\n# Learned attention weights, initialized at uniform 25%\nfor w, lm in zip(weights, landmarks):\n    print(f'{lm} weight: {(w*100):.1f}%')","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:11:29.787704Z","iopub.execute_input":"2023-03-26T14:11:29.788330Z","iopub.status.idle":"2023-03-26T14:11:31.556167Z","shell.execute_reply.started":"2023-03-26T14:11:29.788289Z","shell.execute_reply":"2023-03-26T14:11:31.554824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now that we have the first model ready, let's also work on the second model. ","metadata":{}},{"cell_type":"markdown","source":"# Model - 2 (Linear layer, BN, ReLU)\n> From https://www.kaggle.com/code/roberthatch/gislr-lb-0-63-on-the-shoulders","metadata":{}},{"cell_type":"code","source":"def read_parquet(path): return pd.read_parquet(path)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:11:38.043641Z","iopub.execute_input":"2023-03-26T14:11:38.044879Z","iopub.status.idle":"2023-03-26T14:11:39.772020Z","shell.execute_reply.started":"2023-03-26T14:11:38.044833Z","shell.execute_reply":"2023-03-26T14:11:39.770768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_map = json.load(open(LABEL_MAP_PATH, 'r'))","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:11:39.774394Z","iopub.execute_input":"2023-03-26T14:11:39.779245Z","iopub.status.idle":"2023-03-26T14:11:41.610287Z","shell.execute_reply.started":"2023-03-26T14:11:39.779200Z","shell.execute_reply":"2023-03-26T14:11:41.609048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LIPS_START = 0\nLEFT_HAND_START = LIPS_IDXS.size\nRIGHT_HAND_START = LEFT_HAND_START + LEFT_HAND_IDXS.size\nPOSE_START = RIGHT_HAND_START + RIGHT_HAND_IDXS.size","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:11:41.612020Z","iopub.execute_input":"2023-03-26T14:11:41.612727Z","iopub.status.idle":"2023-03-26T14:11:43.412315Z","shell.execute_reply.started":"2023-03-26T14:11:41.612679Z","shell.execute_reply":"2023-03-26T14:11:43.410769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LEFT_HAND_OFFSET = 468\nPOSE_OFFSET = LEFT_HAND_OFFSET+21\nRIGHT_HAND_OFFSET = POSE_OFFSET+33\n## average over the entire face\n\nlip_landmarks = [61, 185, 40, 39, 37,  0, 267, 269, 270, 409,\n                 291,146, 91,181, 84, 17, 314, 405, 321, 375, \n                 78, 191, 80, 81, 82, 13, 312, 311, 310, 415, \n                 95, 88, 178, 87, 14,317, 402, 318, 324, 308]\nleft_hand_landmarks = list(range(LEFT_HAND_OFFSET, LEFT_HAND_OFFSET+21))\nright_hand_landmarks = list(range(RIGHT_HAND_OFFSET, RIGHT_HAND_OFFSET+21))\npose_landmarks = list(range(POSE_OFFSET, POSE_OFFSET+33))\n\ncfg.SEGMENTS=3\ncfg.NUM_FRAMES=15\ncfg.DROP_Z=True\ncfg.averaging_sets=[\n        [0, 468],\n        [POSE_OFFSET, 33],\n    ]\ncfg.average_over_pose=True\n\n\npoint_landmarks = lip_landmarks + left_hand_landmarks+ right_hand_landmarks\nif not cfg.average_over_pose: \n    point_landmarks = point_landmarks + pose_landmarks\n\n\nTOT_LANDMARKS = len(point_landmarks) + len(cfg.averaging_sets)\nif cfg.DROP_Z:\n    INPUT_SHAPE = (cfg.NUM_FRAMES,TOT_LANDMARKS*2)\nelse:\n    INPUT_SHAPE = (cfg.NUM_FRAMES,TOT_LANDMARKS*3)\n\nINPUT_SHAPE, TOT_LANDMARKS","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:11:43.415625Z","iopub.execute_input":"2023-03-26T14:11:43.416306Z","iopub.status.idle":"2023-03-26T14:11:45.340182Z","shell.execute_reply.started":"2023-03-26T14:11:43.416239Z","shell.execute_reply":"2023-03-26T14:11:45.338965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tf_nan_mean(x, axis=0):\n    return tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), x), axis=axis) / tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), tf.ones_like(x)), axis=axis)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:11:45.342012Z","iopub.execute_input":"2023-03-26T14:11:45.346404Z","iopub.status.idle":"2023-03-26T14:11:47.110062Z","shell.execute_reply.started":"2023-03-26T14:11:45.346351Z","shell.execute_reply":"2023-03-26T14:11:47.108903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tf_nan_std(x, axis=0):\n    d = x - tf_nan_mean(x, axis=axis)\n    return tf.math.sqrt(tf_nan_mean(d * d, axis=axis))","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:11:47.114740Z","iopub.execute_input":"2023-03-26T14:11:47.117539Z","iopub.status.idle":"2023-03-26T14:11:48.834903Z","shell.execute_reply.started":"2023-03-26T14:11:47.117483Z","shell.execute_reply":"2023-03-26T14:11:48.831572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def flatten_means_and_stds(x, axis=0):\n    # Get means and stds\n    x_mean = tf_nan_mean(x, axis=0)\n    x_std  = tf_nan_std(x,  axis=0)\n\n    x_out = tf.concat([x_mean, x_std], axis=0)\n    x_out = tf.reshape(x_out, (1, INPUT_SHAPE[1]*2))\n    x_out = tf.where(tf.math.is_finite(x_out), x_out, tf.zeros_like(x_out))\n    return x_out","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:11:48.839914Z","iopub.execute_input":"2023-03-26T14:11:48.840256Z","iopub.status.idle":"2023-03-26T14:11:50.583004Z","shell.execute_reply.started":"2023-03-26T14:11:48.840204Z","shell.execute_reply":"2023-03-26T14:11:50.581835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_generation(x_in):\n    if not isinstance(x_in, (np.ndarray, tf.Tensor)): \n        x_in = load_relevant_data_subset(x_in)\n    if cfg.DROP_Z:\n        x_in = x_in[:, :, 0:2]\n    x_list = [tf.expand_dims(tf_nan_mean(x_in[:, av_set[0]:av_set[0]+av_set[1], :], axis=1), axis=1) for av_set in cfg.averaging_sets]\n    x_list.append(tf.gather(x_in, point_landmarks, axis=1))\n    x = tf.concat(x_list, 1)\n\n    x_padded = x\n    \n    for i in range(cfg.SEGMENTS):\n        # once right pad, once left\n        p0 = tf.where( ((tf.shape(x_padded)[0] % cfg.SEGMENTS) > 0) & ((i % 2) != 0) , 1, 0)\n        p1 = tf.where( ((tf.shape(x_padded)[0] % cfg.SEGMENTS) > 0) & ((i % 2) == 0) , 1, 0)\n        paddings = [[p0, p1], [0, 0], [0, 0]]\n        x_padded = tf.pad(x_padded, paddings, mode=\"SYMMETRIC\")\n    x_list = tf.split(x_padded, cfg.SEGMENTS)\n    x_list = [flatten_means_and_stds(_x, axis=0) for _x in x_list]\n\n    x_list.append(flatten_means_and_stds(x, axis=0))\n\n    ## Resize only dimension 0. Resize can't handle nan, so replace nan with that dimension's avg value to reduce impact.\n    x = tf.image.resize(\n        tf.where(tf.math.is_finite(x), x, tf_nan_mean(x, axis=0)), \n        [cfg.NUM_FRAMES, TOT_LANDMARKS])\n    x = tf.reshape(x, (1, INPUT_SHAPE[0]*INPUT_SHAPE[1]))\n    x = tf.where(tf.math.is_nan(x), tf.zeros_like(x), x)\n    x_list.append(x)\n    x = tf.concat(x_list, axis=1)\n    return x","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:11:50.584611Z","iopub.execute_input":"2023-03-26T14:11:50.585172Z","iopub.status.idle":"2023-03-26T14:11:52.583923Z","shell.execute_reply.started":"2023-03-26T14:11:50.585135Z","shell.execute_reply":"2023-03-26T14:11:52.582672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FeatureGen(tf.keras.layers.Layer):\n    def __init__(self):\n        super(FeatureGen, self).__init__()\n    \n    def call(self, x_in):\n        return feature_generation(x_in)\n\nfeature_gen = FeatureGen()\nfeature_gen(load_relevant_data_subset(train.file_path[0])).shape","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:11:52.585609Z","iopub.execute_input":"2023-03-26T14:11:52.586312Z","iopub.status.idle":"2023-03-26T14:11:54.377953Z","shell.execute_reply.started":"2023-03-26T14:11:52.586267Z","shell.execute_reply":"2023-03-26T14:11:54.376665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now that we have defined the preprocessing and required utils, let's load the `X` and `y` data.","metadata":{}},{"cell_type":"code","source":"X_PATH = '/kaggle/input/asl-features/feature-set-one/feature_data.npy'\nY_PATH = '/kaggle/input/asl-features/feature-set-one/feature_labels.npy'","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:12:08.884532Z","iopub.execute_input":"2023-03-26T14:12:08.885480Z","iopub.status.idle":"2023-03-26T14:12:10.743700Z","shell.execute_reply.started":"2023-03-26T14:12:08.885441Z","shell.execute_reply":"2023-03-26T14:12:10.742460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nX = np.load(X_PATH)\ny = np.load(Y_PATH)\nX.shape, y.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:12:10.750844Z","iopub.execute_input":"2023-03-26T14:12:10.752862Z","iopub.status.idle":"2023-03-26T14:12:22.497362Z","shell.execute_reply.started":"2023-03-26T14:12:10.752817Z","shell.execute_reply":"2023-03-26T14:12:22.495672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"> As per the original notebook, in this case we do-not use $Z$ axis data.","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(TRAIN_CSV_PATH)\nlabel_map = json.load(open(LABEL_MAP_PATH, 'r'))\ntrain_df['label'] = train_df.sign.map(label_map)\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:12:34.653420Z","iopub.execute_input":"2023-03-26T14:12:34.654123Z","iopub.status.idle":"2023-03-26T14:12:36.611833Z","shell.execute_reply.started":"2023-03-26T14:12:34.654085Z","shell.execute_reply":"2023-03-26T14:12:36.610451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"N_PARTICIPANTS = train_df.participant_id.nunique()\nsgkf = StratifiedGroupKFold(n_splits=7, shuffle=True, random_state=43)\ntrain_df['fold'] = -1\nfor i, (train_idx, val_idx) in enumerate(sgkf.split(train_df.index, train_df.sign, train_df.participant_id)):\n    train_df.loc[val_idx, 'fold'] = i\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:12:36.617982Z","iopub.execute_input":"2023-03-26T14:12:36.620691Z","iopub.status.idle":"2023-03-26T14:12:38.773840Z","shell.execute_reply.started":"2023-03-26T14:12:36.620650Z","shell.execute_reply":"2023-03-26T14:12:38.773022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check ordering is right\nnp.all(np.equal(train_df.label.values, y))","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:12:38.777776Z","iopub.execute_input":"2023-03-26T14:12:38.778549Z","iopub.status.idle":"2023-03-26T14:12:40.628403Z","shell.execute_reply.started":"2023-03-26T14:12:38.778488Z","shell.execute_reply":"2023-03-26T14:12:40.627267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_idxs = train_df.query(\"fold!=0\").index.values\nval_idxs = train_df.query(\"fold==0\").index.values\n\nx_train, y_train = X[train_idxs], y[train_idxs]\nx_test, y_test = X[val_idxs], y[val_idxs]\n\nn_classes = train_df.sign.nunique() #250","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:12:40.634643Z","iopub.execute_input":"2023-03-26T14:12:40.635833Z","iopub.status.idle":"2023-03-26T14:12:43.591886Z","shell.execute_reply.started":"2023-03-26T14:12:40.635791Z","shell.execute_reply":"2023-03-26T14:12:43.590628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train.shape, x_test.shape, y_train.shape, y_test.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:12:43.593371Z","iopub.execute_input":"2023-03-26T14:12:43.594094Z","iopub.status.idle":"2023-03-26T14:12:45.356908Z","shell.execute_reply.started":"2023-03-26T14:12:43.594043Z","shell.execute_reply":"2023-03-26T14:12:45.355711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg.BLOCK_SIZES = [2048, 1024, 512]\ncfg.FLAT_FRAME_SHAPE = x_train.shape[1]\ncfg.DROPOUTS = [0.4, 0.2, 0.1]\ncfg.LEARNING_RATE = 1e-3\ncfg.training_data_path = (X_PATH, Y_PATH)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:12:45.358383Z","iopub.execute_input":"2023-03-26T14:12:45.359125Z","iopub.status.idle":"2023-03-26T14:12:47.073752Z","shell.execute_reply.started":"2023-03-26T14:12:45.359081Z","shell.execute_reply":"2023-03-26T14:12:47.072538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cfg.EPOCHS = 25\ncfg.BATCH_SIZE = 512\n\ndecay_steps = ((len(x_train)//cfg.BATCH_SIZE)*cfg.EPOCHS)\n\ncosine_decay_scheduler = tf.keras.optimizers.schedules.CosineDecay(\n    initial_learning_rate = cfg.LEARNING_RATE,\n    decay_steps = decay_steps,\n    alpha=0.001\n)\noptimizer = tf.keras.optimizers.Adam(cosine_decay_scheduler)\n\n\ndef get_lr_metric(optimizer):\n    def lr(y_true, y_pred):\n        return optimizer.lr\n    return lr\n\n\ndef fc_block(inputs, output_channels, dropout=0.2):\n    x = tf.keras.layers.Dense(output_channels)(inputs)\n    x = tf.keras.layers.BatchNormalization()(x)\n    x = tf.keras.layers.Activation(\"gelu\")(x)\n    x = tf.keras.layers.Dropout(dropout)(x)\n    return x\n\ndef get_model(n_labels=250, flat_frame_len=cfg.FLAT_FRAME_SHAPE):\n    _inputs = tf.keras.layers.Input(shape=(flat_frame_len,))\n    x = _inputs\n    \n    # Define layers\n    for i in range(len(cfg.DROPOUTS)):\n        x = fc_block(\n            x, output_channels=cfg.BLOCK_SIZES[i], \n            dropout=cfg.DROPOUTS[i]\n        )\n    \n    # Define output layer\n    _outputs = tf.keras.layers.Dense(n_labels, activation=\"softmax\")(x)\n    \n    # Build the model\n    model = tf.keras.models.Model(inputs=_inputs, outputs=_outputs)\n    return model\n\nlr_metric = get_lr_metric(optimizer)\nmodel_two = get_model()\nmodel_two.compile(optimizer, \"sparse_categorical_crossentropy\", \n              metrics=[\"acc\", lr_metric])\nmodel_two.summary()","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:12:47.075213Z","iopub.execute_input":"2023-03-26T14:12:47.075919Z","iopub.status.idle":"2023-03-26T14:12:49.063047Z","shell.execute_reply.started":"2023-03-26T14:12:47.075877Z","shell.execute_reply":"2023-03-26T14:12:49.061976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow import keras\nimport tensorflow as tf\nfrom tensorflow.keras import layers","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:12:49.068201Z","iopub.execute_input":"2023-03-26T14:12:49.074774Z","iopub.status.idle":"2023-03-26T14:12:50.201305Z","shell.execute_reply.started":"2023-03-26T14:12:49.074733Z","shell.execute_reply":"2023-03-26T14:12:50.200081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if cfg.TRAIN_MODEL:\n    wandb.init(project='kaggle-asl-signs', config=cfg, tags=['model-two', 'final-model', 'conv'])\n    callbacks = [wandb.keras.WandbCallback(monitor='acc', save_model=False)]","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:13:31.569082Z","iopub.execute_input":"2023-03-26T14:13:31.569474Z","iopub.status.idle":"2023-03-26T14:13:33.545716Z","shell.execute_reply.started":"2023-03-26T14:13:31.569439Z","shell.execute_reply":"2023-03-26T14:13:33.544553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if cfg.TRAIN_MODEL:\n    history = model_two.fit(\n        X, y,\n        epochs=config['EPOCHS'], \n        batch_size=config['BATCH_SIZE'],\n        callbacks=callbacks, \n    )","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:13:33.888065Z","iopub.execute_input":"2023-03-26T14:13:33.888442Z","iopub.status.idle":"2023-03-26T14:13:35.790732Z","shell.execute_reply.started":"2023-03-26T14:13:33.888409Z","shell.execute_reply":"2023-03-26T14:13:35.789475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if cfg.TRAIN_MODEL:\n    model_two.save('./final_model_two')\n    model_two.save_weights('./final_model_two_weights')","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:13:35.792899Z","iopub.execute_input":"2023-03-26T14:13:35.793585Z","iopub.status.idle":"2023-03-26T14:13:37.540321Z","shell.execute_reply.started":"2023-03-26T14:13:35.793535Z","shell.execute_reply":"2023-03-26T14:13:37.538904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if cfg.TRAIN_MODEL:\n    artifact = wandb.Artifact('final_model_two', type='model')\n    artifact.add_file('./final_model_two_weights.data-00000-of-00001')\n    artifact.add_file('./final_model_two_weights.index')\n    run.log_artifact(artifact)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:13:37.543213Z","iopub.execute_input":"2023-03-26T14:13:37.547400Z","iopub.status.idle":"2023-03-26T14:13:39.333220Z","shell.execute_reply.started":"2023-03-26T14:13:37.546749Z","shell.execute_reply":"2023-03-26T14:13:39.331886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# download model\ntry:\n    artifact = run.use_artifact('amanarora/asl-sings/final_model_two:v1', type='model')\nexcept NameError:\n    run = wandb.init(project=\"kaggle-asl-signs\")\n    artifact = run.use_artifact('amanarora/asl-sings/final_model_two:v1', type='model')\nartifact_dir = artifact.download()\n!ls {artifact_dir}","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:13:39.335324Z","iopub.execute_input":"2023-03-26T14:13:39.336050Z","iopub.status.idle":"2023-03-26T14:13:50.512146Z","shell.execute_reply.started":"2023-03-26T14:13:39.336007Z","shell.execute_reply":"2023-03-26T14:13:50.510860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_two.load_weights(f\"{artifact_dir}/final_model_two_weights\")\ny_val_pred = model_two.predict(x_test).argmax(axis=1)\nnp.mean(y_test == y_val_pred)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:14:15.412958Z","iopub.execute_input":"2023-03-26T14:14:15.413460Z","iopub.status.idle":"2023-03-26T14:14:19.117301Z","shell.execute_reply.started":"2023-03-26T14:14:15.413416Z","shell.execute_reply":"2023-03-26T14:14:19.116164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Great, now we have two models, the first model has about 99% accuracy (remeber the val is part of the actual training set, so it's only a sanity check), and the second model is 82.5%. What does this mean? The first model definitely fits to the data better. \n\n> It also means that it's easier to overfit model-one compared to model-two.","metadata":{}},{"cell_type":"markdown","source":"# Create Final Inference Model & TFLITE ","metadata":{}},{"cell_type":"markdown","source":"Let's now create the final model that uses both the models - `model_one` and `model_two`. Remember model-1 is transformer based and model-2 is a simple fully-connected model. ","metadata":{}},{"cell_type":"code","source":"class FinalModel(tf.keras.Model):\n    def __init__(self, model_1, model_2, pp_layer_1, pp_layer_2):\n        super().__init__()\n        self.model_1 = model_1\n        self.model_2 = model_2\n        self.pp_layer_1 = pp_layer_1\n        self.pp_layer_2 = pp_layer_2\n\n    @tf.function(input_signature=[tf.TensorSpec(shape=[None, cfg.N_ROWS, cfg.N_DIMS], dtype=tf.float32, name='inputs')])        \n    def __call__(self, inputs):\n        #model-2 (transformer)\n        x, non_empty_frame_idxs = self.pp_layer_2(inputs)\n        x = tf.expand_dims(x, axis=0)\n        non_empty_frame_idxs = tf.expand_dims(non_empty_frame_idxs, axis=0)\n        _outputs_2 = self.model_2({ 'frames': x, 'non_empty_frame_idxs': non_empty_frame_idxs })\n        \n        # model-1 (custom)\n        x = self.pp_layer_1(tf.cast(inputs, dtype=tf.float32))\n        _outputs_1 = tf.expand_dims(self.model_1(x)[0, :], axis=0)\n        \n        outputs = tf.math.reduce_mean(tf.concat([_outputs_1, _outputs_2], axis=0), axis=0)\n        \n        # Return a dictionary with the output tensor\n        return {'outputs': outputs}        ","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:15:22.741272Z","iopub.execute_input":"2023-03-26T14:15:22.741901Z","iopub.status.idle":"2023-03-26T14:15:24.579916Z","shell.execute_reply.started":"2023-03-26T14:15:22.741847Z","shell.execute_reply":"2023-03-26T14:15:24.578758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model = FinalModel(model_two, model_one, feature_generation, preprocess_layer)","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:15:24.724656Z","iopub.execute_input":"2023-03-26T14:15:24.725781Z","iopub.status.idle":"2023-03-26T14:15:26.589479Z","shell.execute_reply.started":"2023-03-26T14:15:24.725728Z","shell.execute_reply":"2023-03-26T14:15:26.588236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"demo_raw_data = load_relevant_data_subset(train['file_path'].values[1])\ndemo_output = final_model(demo_raw_data)[\"outputs\"]\ndemo_prediction = demo_output.numpy().argmax()\nprint(f'demo_prediction: {demo_prediction}, correct: {train.iloc[1][\"sign_ord\"]}')","metadata":{"execution":{"iopub.status.busy":"2023-03-26T14:15:26.591486Z","iopub.execute_input":"2023-03-26T14:15:26.592382Z","iopub.status.idle":"2023-03-26T14:15:30.487889Z","shell.execute_reply.started":"2023-03-26T14:15:26.592341Z","shell.execute_reply":"2023-03-26T14:15:30.486820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"converter = tf.lite.TFLiteConverter.from_keras_model(final_model)\nconverter.optimizations = [tf.lite.Optimize.DEFAULT]\ntflite_model = converter.convert()\nwith open('./model.tflite', 'wb') as f:\n    f.write(tflite_model)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!du -sh ./model.tflite","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!zip submission.zip ./model.tflite","metadata":{},"execution_count":null,"outputs":[]}]}