{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Data Analysys and Exploration**","metadata":{}},{"cell_type":"code","source":"#Importing nessecassary libraries for reading and mapulating data our dataset\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\ndf=pd.read_csv(\"/kaggle/input/asl-signs/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:14.431902Z","iopub.execute_input":"2023-05-03T14:29:14.432784Z","iopub.status.idle":"2023-05-03T14:29:14.533535Z","shell.execute_reply.started":"2023-05-03T14:29:14.432719Z","shell.execute_reply":"2023-05-03T14:29:14.532440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('We have {} entries(Videos) and {} columns'.format(df.shape[0],df.shape[1]))\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:14.536182Z","iopub.execute_input":"2023-05-03T14:29:14.536591Z","iopub.status.idle":"2023-05-03T14:29:14.542963Z","shell.execute_reply.started":"2023-05-03T14:29:14.536550Z","shell.execute_reply":"2023-05-03T14:29:14.541800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head() # displaying the first 5 elements in the data\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:14.544649Z","iopub.execute_input":"2023-05-03T14:29:14.545357Z","iopub.status.idle":"2023-05-03T14:29:14.560556Z","shell.execute_reply.started":"2023-05-03T14:29:14.545289Z","shell.execute_reply":"2023-05-03T14:29:14.559410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info() # The column details of the data: dtype,the number of no-null values\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:14.563197Z","iopub.execute_input":"2023-05-03T14:29:14.563784Z","iopub.status.idle":"2023-05-03T14:29:14.588663Z","shell.execute_reply.started":"2023-05-03T14:29:14.563747Z","shell.execute_reply":"2023-05-03T14:29:14.587632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('We have {} different signs'.format(len(df.sign.unique())))# The number of different signs that we have\nprint(df.sign.unique())\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:14.589939Z","iopub.execute_input":"2023-05-03T14:29:14.590182Z","iopub.status.idle":"2023-05-03T14:29:14.610967Z","shell.execute_reply.started":"2023-05-03T14:29:14.590158Z","shell.execute_reply":"2023-05-03T14:29:14.609884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p=pd.DataFrame(df.sign.value_counts())#The number of vidoes which collected for each sign\np.columns=['Number_of_vidoes']\np","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:14.612350Z","iopub.execute_input":"2023-05-03T14:29:14.612870Z","iopub.status.idle":"2023-05-03T14:29:14.633763Z","shell.execute_reply.started":"2023-05-03T14:29:14.612835Z","shell.execute_reply":"2023-05-03T14:29:14.632561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:14.635492Z","iopub.execute_input":"2023-05-03T14:29:14.635856Z","iopub.status.idle":"2023-05-03T14:29:14.641544Z","shell.execute_reply.started":"2023-05-03T14:29:14.635821Z","shell.execute_reply":"2023-05-03T14:29:14.640170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('We have {} participants'.format(len(df.participant_id.unique()))) #Number of participants\ndf.participant_id.unique() # To display the different participants\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:14.643270Z","iopub.execute_input":"2023-05-03T14:29:14.643636Z","iopub.status.idle":"2023-05-03T14:29:14.658670Z","shell.execute_reply.started":"2023-05-03T14:29:14.643599Z","shell.execute_reply":"2023-05-03T14:29:14.657324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indexs=[0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20]\nk=pd.DataFrame(df.participant_id.value_counts())\nk.columns=['Created Videos']\nk.index\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:14.660075Z","iopub.execute_input":"2023-05-03T14:29:14.660820Z","iopub.status.idle":"2023-05-03T14:29:14.671709Z","shell.execute_reply.started":"2023-05-03T14:29:14.660784Z","shell.execute_reply":"2023-05-03T14:29:14.670549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import libraries\nimport numpy as np\nimport matplotlib.pyplot as plt\n \n # Creating dataset\nparticipant=k.index\n \ndata = k['Created Videos']\n \n # Creating explode data\nexplode = (0.0, 0.0, 0.0, 0.0, 0.0, 0.0,0.0,0.0,0.0,\n          0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,0.0,\n          0.0)\n \n# Creating color parameters\ncolors = ( \"orange\", \"cyan\", \"brown\",\n          \"grey\", \"indigo\", \"beige\")\n \n# Wedge properties\nwp = { 'linewidth' : 1, 'edgecolor' : \"green\" }\n \n# Creating autocpt arguments\ndef func(pct, allvalues):\n    absolute = int(pct / 100.*np.sum(allvalues))\n    return \"{:.1f}%\\n({:d} )\".format(pct, absolute)\n \n# Creating plot\nfig, ax = plt.subplots(figsize =(10, 7))\nwedges, texts, autotexts = ax.pie(data,\n                                  autopct = lambda pct: func(pct, data),\n                                  explode = explode,\n                                  labels = participant,\n                                  shadow = True,\n                                  colors = colors,\n                                  startangle = 90,\n                                  wedgeprops = wp,\n                                  textprops = dict(color =\"magenta\"))\n \n# Adding legend\nax.legend(wedges, participant,\n          title =\"Participant\",\n          loc =\"center left\",\n          bbox_to_anchor =(1, 0, 0.5, 1))\n \nplt.setp(autotexts, size = 8, weight =\"bold\")\nax.set_title(\"Participant contribution\")\n \n# show plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:14.676773Z","iopub.execute_input":"2023-05-03T14:29:14.677407Z","iopub.status.idle":"2023-05-03T14:29:15.285493Z","shell.execute_reply.started":"2023-05-03T14:29:14.677380Z","shell.execute_reply":"2023-05-03T14:29:15.284453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# concatenating the path to file with the video and \n#path to vidoe and which sign it is\npath='/kaggle/input/asl-signs/'\nfiles=np.array(path+df.path+'@'+df.sign)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:15.286926Z","iopub.execute_input":"2023-05-03T14:29:15.287981Z","iopub.status.idle":"2023-05-03T14:29:15.338891Z","shell.execute_reply.started":"2023-05-03T14:29:15.287940Z","shell.execute_reply":"2023-05-03T14:29:15.337774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Video Preprocessing**","metadata":{}},{"cell_type":"code","source":"videos=[]\nsign=[]\nfor i in files:\n    splited=i.split(\"@\")\n    videos.append(splited[0])# storing the created paths in a array\n    sign.append(splited[1])  # storing the relavant sign to that video\n\n    \nprint(len(videos)) \nprint(len(sign))\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:15.343367Z","iopub.execute_input":"2023-05-03T14:29:15.346206Z","iopub.status.idle":"2023-05-03T14:29:15.415009Z","shell.execute_reply.started":"2023-05-03T14:29:15.346164Z","shell.execute_reply":"2023-05-03T14:29:15.413861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"video=pd.read_parquet(videos[0])\nvideo\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:15.416489Z","iopub.execute_input":"2023-05-03T14:29:15.416864Z","iopub.status.idle":"2023-05-03T14:29:15.445932Z","shell.execute_reply.started":"2023-05-03T14:29:15.416828Z","shell.execute_reply":"2023-05-03T14:29:15.444827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The datatypes and more information of the columns in the data set\nvideo.info()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:15.447383Z","iopub.execute_input":"2023-05-03T14:29:15.447761Z","iopub.status.idle":"2023-05-03T14:29:15.462673Z","shell.execute_reply.started":"2023-05-03T14:29:15.447722Z","shell.execute_reply":"2023-05-03T14:29:15.461530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The number of frames that we have\nprint('There are {} frames'.format(len(video.frame.unique())))\n#Displaying the frames \nvideo.frame.unique()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:15.464484Z","iopub.execute_input":"2023-05-03T14:29:15.464953Z","iopub.status.idle":"2023-05-03T14:29:15.475295Z","shell.execute_reply.started":"2023-05-03T14:29:15.464900Z","shell.execute_reply":"2023-05-03T14:29:15.474127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Displaying the type of body expressions \n# or part that we have in the video\nprint('There are {} body parts/expressions'.format(len(video.type.unique())))\n\nvideo.type.unique()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:15.476713Z","iopub.execute_input":"2023-05-03T14:29:15.477651Z","iopub.status.idle":"2023-05-03T14:29:15.492032Z","shell.execute_reply.started":"2023-05-03T14:29:15.477606Z","shell.execute_reply":"2023-05-03T14:29:15.490954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Take one frame as a sample to sketch it \ndframe=video.loc[video['frame']==20]\ndframe\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:15.493614Z","iopub.execute_input":"2023-05-03T14:29:15.494897Z","iopub.status.idle":"2023-05-03T14:29:15.519055Z","shell.execute_reply.started":"2023-05-03T14:29:15.494869Z","shell.execute_reply":"2023-05-03T14:29:15.518016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Extracting the face, right-hand,lefthand and pose into a dictionary\ndface=dframe.loc[dframe['type']=='face']\ndright=dframe.loc[dframe['type']=='right_hand']\ndpose=dframe.loc[dframe['type']=='pose']\ndleft=dframe.loc[dframe['type']=='left_hand']\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:15.521407Z","iopub.execute_input":"2023-05-03T14:29:15.522114Z","iopub.status.idle":"2023-05-03T14:29:15.530367Z","shell.execute_reply.started":"2023-05-03T14:29:15.522066Z","shell.execute_reply":"2023-05-03T14:29:15.529082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Skectch the face for frame 20\nimport matplotlib.pyplot as plt\nfrom mpl_toolkits.mplot3d import Axes3D\nimport numpy as np\nfig=plt.figure()\nax=fig.add_subplot(111,projection='3d')\nax.scatter(dface.x,dface.y)\nax.view_init(elev=90,azim=90)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:15.532132Z","iopub.execute_input":"2023-05-03T14:29:15.532565Z","iopub.status.idle":"2023-05-03T14:29:15.754489Z","shell.execute_reply.started":"2023-05-03T14:29:15.532503Z","shell.execute_reply":"2023-05-03T14:29:15.753452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:15.756231Z","iopub.execute_input":"2023-05-03T14:29:15.756886Z","iopub.status.idle":"2023-05-03T14:29:15.777336Z","shell.execute_reply.started":"2023-05-03T14:29:15.756845Z","shell.execute_reply":"2023-05-03T14:29:15.776222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"videos=[]\nsign=[]\nfor i in files:\n    splited=i.split(\"@\")\n    videos.append(splited[0])\n    sign.append(splited[1])\nprint(len(videos))\nprint(len(sign))\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:15.779100Z","iopub.execute_input":"2023-05-03T14:29:15.779585Z","iopub.status.idle":"2023-05-03T14:29:15.853722Z","shell.execute_reply.started":"2023-05-03T14:29:15.779536Z","shell.execute_reply":"2023-05-03T14:29:15.852578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"video=pd.read_parquet(videos[0])\nvideo","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:15.855223Z","iopub.execute_input":"2023-05-03T14:29:15.855590Z","iopub.status.idle":"2023-05-03T14:29:15.881811Z","shell.execute_reply.started":"2023-05-03T14:29:15.855551Z","shell.execute_reply":"2023-05-03T14:29:15.880719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyarrow","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:15.883574Z","iopub.execute_input":"2023-05-03T14:29:15.883945Z","iopub.status.idle":"2023-05-03T14:29:25.611570Z","shell.execute_reply.started":"2023-05-03T14:29:15.883908Z","shell.execute_reply":"2023-05-03T14:29:25.610236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install imageio","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:25.613572Z","iopub.execute_input":"2023-05-03T14:29:25.614263Z","iopub.status.idle":"2023-05-03T14:29:35.428780Z","shell.execute_reply.started":"2023-05-03T14:29:25.614213Z","shell.execute_reply":"2023-05-03T14:29:35.427533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pyarrow.parquet as pq\nimport imageio\nimport cv2\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:35.432812Z","iopub.execute_input":"2023-05-03T14:29:35.433167Z","iopub.status.idle":"2023-05-03T14:29:35.438929Z","shell.execute_reply.started":"2023-05-03T14:29:35.433132Z","shell.execute_reply":"2023-05-03T14:29:35.437855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The datatypes and more information of the columns in the data set\nvideo.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:35.440087Z","iopub.execute_input":"2023-05-03T14:29:35.440506Z","iopub.status.idle":"2023-05-03T14:29:35.463054Z","shell.execute_reply.started":"2023-05-03T14:29:35.440461Z","shell.execute_reply":"2023-05-03T14:29:35.462096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The number of frames that we have\nprint('There are {} frames'.format(len(video.frame.unique())))","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:35.466435Z","iopub.execute_input":"2023-05-03T14:29:35.466735Z","iopub.status.idle":"2023-05-03T14:29:35.473708Z","shell.execute_reply.started":"2023-05-03T14:29:35.466707Z","shell.execute_reply":"2023-05-03T14:29:35.472512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"video","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:35.487794Z","iopub.execute_input":"2023-05-03T14:29:35.488074Z","iopub.status.idle":"2023-05-03T14:29:35.508852Z","shell.execute_reply.started":"2023-05-03T14:29:35.488047Z","shell.execute_reply":"2023-05-03T14:29:35.507720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport tensorflow as tf\nimport pyarrow.parquet as pq\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport os\nimport glob\nimport tqdm\nimport random\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom matplotlib import pyplot as plt\nimport struct\nfrom tensorflow.python.summary.summary_iterator import summary_iterator\n\ndf=pd.read_csv('/kaggle/input/asl-signs/train.csv')\nlandmark_files_loc = \"/kaggle/input/asl-signs/train_landmark_files\"\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:35.510293Z","iopub.execute_input":"2023-05-03T14:29:35.510991Z","iopub.status.idle":"2023-05-03T14:29:35.657124Z","shell.execute_reply.started":"2023-05-03T14:29:35.510948Z","shell.execute_reply":"2023-05-03T14:29:35.655950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '/kaggle/input/asl-signs'\ndf[\"path\"] = data_dir +\"/\"+ df[\"path\"]\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:35.662201Z","iopub.execute_input":"2023-05-03T14:29:35.666851Z","iopub.status.idle":"2023-05-03T14:29:35.702305Z","shell.execute_reply.started":"2023-05-03T14:29:35.666803Z","shell.execute_reply":"2023-05-03T14:29:35.701184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\ndef read_json(path):\n    with open(path, \"r\") as file:\n        json_data = json.load(file)\n    return json_data\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:35.703717Z","iopub.execute_input":"2023-05-03T14:29:35.704457Z","iopub.status.idle":"2023-05-03T14:29:35.716449Z","shell.execute_reply.started":"2023-05-03T14:29:35.704418Z","shell.execute_reply":"2023-05-03T14:29:35.715247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#to attach label for each video into our data set\ns2p_map = read_json('/kaggle/input/asl-signs/sign_to_prediction_index_map.json')\np2s_map = {j: k for k, j in s2p_map.items()}\n\nencoder = lambda x: s2p_map.get(x)\ndecoder = lambda x: p2s_map.get(x)\n\ndf[\"label\"] = df[\"sign\"].map(encoder)\ndf.head(7)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:35.717912Z","iopub.execute_input":"2023-05-03T14:29:35.718474Z","iopub.status.idle":"2023-05-03T14:29:35.810698Z","shell.execute_reply.started":"2023-05-03T14:29:35.718431Z","shell.execute_reply":"2023-05-03T14:29:35.809722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#extract our features\ntrain_df=df\nclass FeatureGen(nn.Module):\n    def __init__(self):\n        super(FeatureGen, self).__init__()\n\n    def forward(self, x):\n        x = torch.from_numpy(x)\n        x = torch.where(torch.isnan(x), torch.zeros_like(x), x)\n        x = torch.mean(x, axis=0)\n        return x\n    \nfeature_converter = FeatureGen()\n#get the length of the data \ndata_length = (len(train_df))\ndata_lenght_experiment = int(len(train_df)/10)\n\n#extract our xyz data from each and every video and put them in a form of an array for each video separatetly\ndef load_relevant_data_subset(pq_path):\n    data_columns = [\"x\", \"y\", \"z\"]\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)\ndef convert_row(row):\n    x = load_relevant_data_subset(os.path.join(\"/kaggle/input/asl-signs\", row.path))\n    x = feature_converter(x)\n    return x, row.label\n\nfrom tqdm import tqdm\n\nROWS_PER_FRAME = 543\n#convert the extracted data into an npy array for features and for labels\ndef convert_and_save_data():\n    np_features = np.zeros((data_lenght_experiment, ROWS_PER_FRAME, 3))\n    np_labels = np.zeros(data_lenght_experiment)\n\n    for index, row in tqdm(train_df.iterrows()):\n        if index > data_lenght_experiment - 1:\n            break\n\n        data = load_relevant_data_subset(row.path)\n        feature, label = convert_row(row)\n        np_features[index, :, :] = feature.to(torch.double)\n        np_labels[index] = label\n\n    np.save(\"features.npy\", np_features)\n    np.save(\"labels.npy\", np_labels)    \n    \n   ","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:35.814928Z","iopub.execute_input":"2023-05-03T14:29:35.817235Z","iopub.status.idle":"2023-05-03T14:29:35.834463Z","shell.execute_reply.started":"2023-05-03T14:29:35.817197Z","shell.execute_reply":"2023-05-03T14:29:35.833486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# access our kaggle working output\n!rm -rf /kaggle/working/*.npy\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:35.838866Z","iopub.execute_input":"2023-05-03T14:29:35.841904Z","iopub.status.idle":"2023-05-03T14:29:37.165178Z","shell.execute_reply.started":"2023-05-03T14:29:35.841856Z","shell.execute_reply":"2023-05-03T14:29:37.163201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#upload our converted data into kaggle output \ntry:\n    features = torch.from_numpy(np.load(\"/kaggle/working/features.npy\")).float()\n    labels = torch.from_numpy(np.load(\"/kaggle/working/labels.npy\")).long()\nexcept:\n    convert_and_save_data()\nfinally:\n    features = torch.from_numpy(np.load(\"/kaggle/working/features.npy\")).float()\n    labels = torch.from_numpy(np.load(\"/kaggle/working/labels.npy\")).long()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:29:37.168190Z","iopub.execute_input":"2023-05-03T14:29:37.169375Z","iopub.status.idle":"2023-05-03T14:31:53.274753Z","shell.execute_reply.started":"2023-05-03T14:29:37.169330Z","shell.execute_reply":"2023-05-03T14:31:53.273680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#splitting test and train datasets\nfrom sklearn.model_selection import train_test_split\n\nX_train, X_val, y_train, y_val = train_test_split(\n    features, labels, test_size=0.2, stratify=labels, random_state=42\n)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:31:53.276479Z","iopub.execute_input":"2023-05-03T14:31:53.276982Z","iopub.status.idle":"2023-05-03T14:31:53.306771Z","shell.execute_reply.started":"2023-05-03T14:31:53.276940Z","shell.execute_reply":"2023-05-03T14:31:53.305779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Dataset(torch.utils.data.Dataset):\n    def __init__(self, X, y):\n        self.X = X\n        self.y = y\n\n    def __len__(self):\n        return len(self.y)\n\n    def __getitem__(self, i):\n        return self.X[i], self.y[i]","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:31:53.308122Z","iopub.execute_input":"2023-05-03T14:31:53.309009Z","iopub.status.idle":"2023-05-03T14:31:53.315980Z","shell.execute_reply.started":"2023-05-03T14:31:53.308964Z","shell.execute_reply":"2023-05-03T14:31:53.314818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#extracting our train and test dataset\ntrain_dataset = Dataset(X_train, y_train)\nval_dataset = Dataset(X_val, y_val)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:31:53.317471Z","iopub.execute_input":"2023-05-03T14:31:53.318114Z","iopub.status.idle":"2023-05-03T14:31:53.326541Z","shell.execute_reply.started":"2023-05-03T14:31:53.318033Z","shell.execute_reply":"2023-05-03T14:31:53.325563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataloader = torch.utils.data.DataLoader(train_dataset, batch_size=128, \n                                               shuffle=True,num_workers=1,\n                                               pin_memory=True, drop_last=True)\nval_dataloader = torch.utils.data.DataLoader(train_dataset, batch_size=128,\n                                             shuffle=False,num_workers=1,\n                                             pin_memory=True, drop_last=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:31:53.328393Z","iopub.execute_input":"2023-05-03T14:31:53.328849Z","iopub.status.idle":"2023-05-03T14:31:53.337361Z","shell.execute_reply.started":"2023-05-03T14:31:53.328812Z","shell.execute_reply":"2023-05-03T14:31:53.336322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nfrom torch.autograd import Variable \n#defining our pytorch model\n\nclass Model(nn.Module):\n    def __init__(self):\n        super(Model, self).__init__()\n        self.layer1 = nn.Linear(3, 128)\n        self.layer2 = nn.Linear(128, 64)\n        self.layer3 = nn.Linear(64, 32)\n        self.layer4 = nn.Linear(32, 16)\n        self.layer5 = nn.Linear(16 * 543, 250)\n        self.relu = nn.ReLU()\n        self.flatten = nn.Flatten()\n\n    def forward(self, x):\n        x = self.relu(self.layer1(x))\n        x = self.relu(self.layer2(x))\n        x = self.relu(self.layer3(x))\n        x = self.relu(self.layer4(x))\n        x = self.flatten(x)\n        x = self.layer5(x)\n        return x\n\nmodel = Model()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:31:53.339135Z","iopub.execute_input":"2023-05-03T14:31:53.339659Z","iopub.status.idle":"2023-05-03T14:31:53.369550Z","shell.execute_reply.started":"2023-05-03T14:31:53.339611Z","shell.execute_reply":"2023-05-03T14:31:53.368668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel.to(device)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:31:53.371203Z","iopub.execute_input":"2023-05-03T14:31:53.371573Z","iopub.status.idle":"2023-05-03T14:31:53.383742Z","shell.execute_reply.started":"2023-05-03T14:31:53.371535Z","shell.execute_reply":"2023-05-03T14:31:53.382350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#learning_rate = 0.0001 \ncriterion = torch.nn.CrossEntropyLoss()    # mean-squared error for regression\noptimizer = torch.optim.Adam(model.parameters(), lr=0.0001) ","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:31:53.385296Z","iopub.execute_input":"2023-05-03T14:31:53.385680Z","iopub.status.idle":"2023-05-03T14:31:53.391601Z","shell.execute_reply.started":"2023-05-03T14:31:53.385642Z","shell.execute_reply":"2023-05-03T14:31:53.390585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_epochs = 40\nfor epoch in range(num_epochs):\n    train_loss, train_correct, train_n, val_loss, val_correct, val_n = 0,0,0,0,0,0\n    model.train()\n    \n    for ibatch, (X, y) in enumerate(train_dataloader):\n        X, y = X.to(device), y.to(device)\n        optimizer.zero_grad()\n        y_pred = model(X)\n        loss = criterion(y_pred, y)\n        \n        train_n += y.size(0)\n        train_loss += loss.item()\n        train_correct += (y_pred.argmax(1) == y).type(torch.float).sum().item()\n\n        loss.backward()\n        optimizer.step()\n        \n    train_loss /= ibatch   \n    train_correct /= train_n\n    \n    model.eval()\n    \n    for ibatch, (X, y) in enumerate(val_dataloader):\n        X, y = X.to(device), y.to(device)\n        with torch.no_grad():\n            y_pred = model(X)\n            loss = criterion(y_pred, y)\n            val_n += y.size(0)\n            val_loss += loss.item()\n            val_correct += (y_pred.argmax(1) == y).type(torch.float).sum().item()\n\n    val_loss /= ibatch   \n    val_correct /= val_n\n    \n    print('Epoch %d/%d loss:%.4f accuracy:%.4f val_loss:%.4f val_accuracy:%.4f' %(epoch + 1, num_epochs, train_loss, train_correct, val_loss, val_correct))\n    train_hist=(epoch + 1, num_epochs, train_loss, train_correct, val_loss, val_correct)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:46:07.178982Z","iopub.execute_input":"2023-05-03T14:46:07.180127Z","iopub.status.idle":"2023-05-03T14:46:43.092260Z","shell.execute_reply.started":"2023-05-03T14:46:07.180076Z","shell.execute_reply":"2023-05-03T14:46:43.090211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.save(model.state_dict(), 'model.pth')\nfrom torchinfo import summary\n\nsummary(model=model, input_size=(128, 543, 3))","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:47:05.826665Z","iopub.execute_input":"2023-05-03T14:47:05.827080Z","iopub.status.idle":"2023-05-03T14:47:05.868055Z","shell.execute_reply.started":"2023-05-03T14:47:05.827043Z","shell.execute_reply":"2023-05-03T14:47:05.867042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls /kaggle/working/asl_sign","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:47:11.483984Z","iopub.execute_input":"2023-05-03T14:47:11.484677Z","iopub.status.idle":"2023-05-03T14:47:12.540660Z","shell.execute_reply.started":"2023-05-03T14:47:11.484614Z","shell.execute_reply":"2023-05-03T14:47:12.539332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q flatbuffers 2> /dev/null\n!pip install -q mediapipe 2> /dev/null","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:37:54.100493Z","iopub.status.idle":"2023-05-03T14:37:54.101651Z","shell.execute_reply.started":"2023-05-03T14:37:54.101342Z","shell.execute_reply":"2023-05-03T14:37:54.101374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"[ROLAND ABEL](http://kaggle kernels pull ted0071/gislr-visualization)","metadata":{}},{"cell_type":"markdown","source":"**Video Animation**","metadata":{}},{"cell_type":"code","source":"#Display video using animation\nimport os\nimport cv2\nimport json\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport mediapipe as mp\nimport matplotlib.pyplot as plt\n\nfrom matplotlib import animation\nfrom pathlib import Path\nimport IPython\nfrom IPython import display\nfrom IPython.display import HTML\n\nimport mediapipe as mp\nfrom mediapipe.framework.formats import landmark_pb2\n\nclass Cfg:\n    RANDOM_STATE = 2023\n    INPUT_ROOT = Path('/kaggle/input/asl-signs/')\n    OUTPUT_ROOT = Path('kaggle/working')\n    INDEX_MAP_FILE = INPUT_ROOT / 'sign_to_prediction_index_map.json'\n    TRAN_FILE = INPUT_ROOT / 'train.csv'\n    INDEX = 'sequence_id'\n    ROW_ID = 'row_id'\n    \ndef create_frames(path, df, height=800, width=800):\n    data = read_landmark_data_by_id(sequence_id, train_data)\n    frame_ids = data['frame'].unique()\n    images = [create_frame(data, frame_id=fid, height=height, width=width) for fid in frame_ids]\n    return np.array(images)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T14:37:54.103486Z","iopub.status.idle":"2023-05-03T14:37:54.104112Z","shell.execute_reply.started":"2023-05-03T14:37:54.103796Z","shell.execute_reply":"2023-05-03T14:37:54.103827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_index_map(file_path=Cfg.INDEX_MAP_FILE):\n    \"\"\"Reads the sign to predict as json file.\"\"\"\n    with open(file_path, \"r\") as f:\n        result = json.load(f)\n    return result    \n\ndef read_train(file_path=Cfg.TRAN_FILE):\n    \"\"\"Reads the train csv as pandas data frame.\"\"\"\n    return pd.read_csv(file_path).set_index(Cfg.INDEX)\n\ndef read_landmark_data_by_path(file_path, input_root=Cfg.INPUT_ROOT):\n    \"\"\"Reads landmak data by the given file path.\"\"\"\n    data = pd.read_parquet(input_root / file_path)\n    return data.set_index(Cfg.ROW_ID)\n\ndef read_landmark_data_by_id(sequence_id, train_data):\n    \"\"\"Reads the landmark data by the given sequence id.\"\"\"\n    file_path = train_data.loc[sequence_id]['path']\n    return read_landmark_data_by_path(file_path)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T15:15:48.204401Z","iopub.execute_input":"2023-05-03T15:15:48.205426Z","iopub.status.idle":"2023-05-03T15:15:48.213815Z","shell.execute_reply.started":"2023-05-03T15:15:48.205387Z","shell.execute_reply":"2023-05-03T15:15:48.212318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = read_train()\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T15:15:55.475334Z","iopub.execute_input":"2023-05-03T15:15:55.475728Z","iopub.status.idle":"2023-05-03T15:15:55.578764Z","shell.execute_reply.started":"2023-05-03T15:15:55.475693Z","shell.execute_reply":"2023-05-03T15:15:55.577507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mp_drawing = mp.solutions.drawing_utils\nmp_hands = mp.solutions.hands\nmp_face_mesh = mp.solutions.face_mesh\nmp_pose = mp.solutions.pose","metadata":{"execution":{"iopub.status.busy":"2023-05-03T15:15:59.365300Z","iopub.execute_input":"2023-05-03T15:15:59.366038Z","iopub.status.idle":"2023-05-03T15:15:59.371536Z","shell.execute_reply.started":"2023-05-03T15:15:59.365998Z","shell.execute_reply":"2023-05-03T15:15:59.370453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_random_sequence_id(train_data):\n    idx = np.random.randint(0, len(train_data))\n    return train_data.index[idx]","metadata":{"execution":{"iopub.status.busy":"2023-05-03T15:16:03.199889Z","iopub.execute_input":"2023-05-03T15:16:03.200866Z","iopub.status.idle":"2023-05-03T15:16:03.206443Z","shell.execute_reply.started":"2023-05-03T15:16:03.200810Z","shell.execute_reply":"2023-05-03T15:16:03.205267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_blank_image(height, width):\n    return np.zeros((height, width, 3), np.uint8)\n\ndef draw_landmarks(\n    data, \n    image, \n    frame_id, \n    landmark_type, \n    connection_type, \n    landmark_color=(255, 0, 0), \n    connection_color=(0, 20, 255), \n    thickness=1, \n    circle_radius=1\n):\n    \"\"\"Draws landmarks\"\"\"\n    df = data.groupby(['frame', 'type']).get_group((frame_id, landmark_type))\n    landmarks = [landmark_pb2.NormalizedLandmark(x=lm.x, y=lm.y, z=lm.z) for idx, lm in df.iterrows()]\n    landmark_list = landmark_pb2.NormalizedLandmarkList(landmark = landmarks)\n\n    mp_drawing.draw_landmarks(\n        image=image,\n        landmark_list=landmark_list, \n        connections=connection_type,\n        landmark_drawing_spec=mp_drawing.DrawingSpec(\n            color=landmark_color, \n            thickness=thickness, \n            circle_radius=circle_radius),\n        connection_drawing_spec=mp_drawing.DrawingSpec(\n            color=connection_color, \n            thickness=thickness, \n            circle_radius=circle_radius))\n    return image\n\ndef draw_left_hand(data, image, frame_id):\n    return draw_landmarks(\n        data, \n        image, \n        frame_id, \n        landmark_type='left_hand', \n        connection_type=mp_hands.HAND_CONNECTIONS,\n        landmark_color=(255, 0, 0),\n        connection_color=(0, 20, 255), \n        thickness=3, \n        circle_radius=3)\n\ndef draw_right_hand(data, image, frame_id):\n    return draw_landmarks(\n        data, \n        image, \n        frame_id, \n        landmark_type='right_hand', \n        connection_type=mp_hands.HAND_CONNECTIONS,\n        landmark_color=(255, 0, 0),\n        connection_color=(0, 20, 255),\n        thickness=3, \n        circle_radius=3)\n\ndef draw_face(data, image, frame_id):\n    return draw_landmarks(\n        data, \n        image, \n        frame_id, \n        landmark_type='face', \n        connection_type=mp_face_mesh.FACEMESH_TESSELATION,\n        landmark_color=(255, 255, 255),\n        connection_color=(0, 255, 0))      \n    \ndef draw_pose(data, image, frame_id):\n    return draw_landmarks(\n        data, \n        image, \n        frame_id, \n        landmark_type='pose', \n        connection_type=mp_pose.POSE_CONNECTIONS,\n        landmark_color=(255, 255, 255),\n        connection_color=(255, 0, 0),\n        thickness=2, \n        circle_radius=2)\n\ndef create_frame(data, frame_id, height=1000, width=1000):\n    image = create_blank_image(height, width)    \n\n    draw_pose(data, image, frame_id) \n    draw_left_hand(data, image, frame_id)    \n    draw_right_hand(data, image, frame_id)  \n    draw_face(data, image, frame_id)\n     \n    return image","metadata":{"execution":{"iopub.status.busy":"2023-05-03T15:16:06.636795Z","iopub.execute_input":"2023-05-03T15:16:06.637742Z","iopub.status.idle":"2023-05-03T15:16:06.655668Z","shell.execute_reply.started":"2023-05-03T15:16:06.637702Z","shell.execute_reply":"2023-05-03T15:16:06.654497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_animation(images, fig, ax):\n    ax.axis('off')\n    \n    ims = []\n    for img in images:\n        im = ax.imshow(img, animated=True)\n        ims.append([im])\n    \n    func_animation = animation.ArtistAnimation(\n        fig, \n        ims, \n        interval=100, \n        blit=True,\n        repeat_delay=1000)\n\n    return func_animation\n\ndef play_animation(path, df, height, width, figsize=(4, 4)):\n    model.predict=df.loc[sequence_id]['sign']\n    frames = create_frames(path, df, height,width)\n    sign = df.loc[sequence_id]['sign']\n    fig, ax = plt.subplots(1, 1, figsize=figsize)\n    anim = create_animation(frames, fig, ax)\n    ax.set_title(f'Sign: {sign}| Predicted:{model.predict}')\n    \n    video = anim.to_html5_video()\n    html = display.HTML(video)\n    display.display(html)\n    plt.close()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T15:24:15.308147Z","iopub.execute_input":"2023-05-03T15:24:15.308556Z","iopub.status.idle":"2023-05-03T15:24:15.318700Z","shell.execute_reply.started":"2023-05-03T15:24:15.308507Z","shell.execute_reply":"2023-05-03T15:24:15.317595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nn=random.randint(0,len(df))\npaths=df.path\ndirectory='/kaggle/input/asl-signs/'\npath=directory+paths[n]\nk=paths[n]\nsequence_id = get_random_sequence_id(train_data)\nplay_animation(sequence_id, train_data,1200, 1200)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T15:21:52.959302Z","iopub.execute_input":"2023-05-03T15:21:52.960029Z","iopub.status.idle":"2023-05-03T15:21:57.540864Z","shell.execute_reply.started":"2023-05-03T15:21:52.959988Z","shell.execute_reply":"2023-05-03T15:21:57.539553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n=random.randint(0,len(df))\npaths=df.path\ndirectory='/kaggle/input/asl-signs/'\npath=directory+paths[n]\nk=paths[n]\nsequence_id = get_random_sequence_id(train_data)\nplay_animation(sequence_id, train_data,1200, 1200)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T15:22:41.396149Z","iopub.execute_input":"2023-05-03T15:22:41.396597Z","iopub.status.idle":"2023-05-03T15:22:47.228892Z","shell.execute_reply.started":"2023-05-03T15:22:41.396555Z","shell.execute_reply":"2023-05-03T15:22:47.227733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n=random.randint(0,len(df))\npaths=df.path\ndirectory='/kaggle/input/asl-signs/'\npath=directory+paths[n]\nk=paths[n]\nsequence_id = get_random_sequence_id(train_data)\nplay_animation(sequence_id, train_data,1200, 1200)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T15:23:00.979507Z","iopub.execute_input":"2023-05-03T15:23:00.980643Z","iopub.status.idle":"2023-05-03T15:23:03.083117Z","shell.execute_reply.started":"2023-05-03T15:23:00.980578Z","shell.execute_reply":"2023-05-03T15:23:03.081972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n=random.randint(0,len(df))\npaths=df.path\ndirectory='/kaggle/input/asl-signs/'\npath=directory+paths[n]\nk=paths[n]\nsequence_id = get_random_sequence_id(train_data)\nplay_animation(sequence_id, train_data,1200, 1200)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T15:23:14.066108Z","iopub.execute_input":"2023-05-03T15:23:14.066543Z","iopub.status.idle":"2023-05-03T15:23:16.147870Z","shell.execute_reply.started":"2023-05-03T15:23:14.066482Z","shell.execute_reply":"2023-05-03T15:23:16.146633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}