{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n\nimport pyarrow.parquet as pq\nimport matplotlib.pyplot as plt\nimport matplotlib.animation as animation\nimport seaborn as sns\nfrom IPython.display import HTML\nimport os\nfrom time import time\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-02T13:58:29.092939Z","iopub.execute_input":"2023-04-02T13:58:29.09367Z","iopub.status.idle":"2023-04-02T13:58:29.630286Z","shell.execute_reply.started":"2023-04-02T13:58:29.093632Z","shell.execute_reply":"2023-04-02T13:58:29.629278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pretty visualization and reflections about models and model input representation\nThis notebook contains visualization and some analysis of asl signs data. See below a models and dataloader","metadata":{}},{"cell_type":"code","source":"data = pq.read_table(\"/kaggle/input/asl-signs/train_landmark_files/16069/100015657.parquet\")\ndata = data.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T13:58:30.703312Z","iopub.execute_input":"2023-04-02T13:58:30.703993Z","iopub.status.idle":"2023-04-02T13:58:30.840959Z","shell.execute_reply.started":"2023-04-02T13:58:30.703935Z","shell.execute_reply":"2023-04-02T13:58:30.839895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.info()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T13:58:31.100133Z","iopub.execute_input":"2023-04-02T13:58:31.102256Z","iopub.status.idle":"2023-04-02T13:58:31.124996Z","shell.execute_reply.started":"2023-04-02T13:58:31.102206Z","shell.execute_reply":"2023-04-02T13:58:31.123725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T13:58:32.107095Z","iopub.execute_input":"2023-04-02T13:58:32.108048Z","iopub.status.idle":"2023-04-02T13:58:32.129389Z","shell.execute_reply.started":"2023-04-02T13:58:32.108004Z","shell.execute_reply":"2023-04-02T13:58:32.128071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's see at frame 103\nface103 = data[data[\"row_id\"].str.contains(\"103-face\")]\nface103","metadata":{"execution":{"iopub.status.busy":"2023-04-02T13:58:32.825544Z","iopub.execute_input":"2023-04-02T13:58:32.826527Z","iopub.status.idle":"2023-04-02T13:58:32.869038Z","shell.execute_reply.started":"2023-04-02T13:58:32.826479Z","shell.execute_reply":"2023-04-02T13:58:32.867844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's visualize keypoints of signs","metadata":{}},{"cell_type":"code","source":"# some functionse to do animation of signs\nface_labels = {label: i for i, label in enumerate(data[\"type\"].unique())}\nprint(face_labels)\nfig, ax = plt.subplots(1, 1, figsize=(10, 10))\nax.set_aspect('equal', 'box')\nax.scatter([], [], s=1)\ndef animate_sequence(frame_idx):\n    ax.cla()\n    ax.set_xlim((0, 1))\n    ax.set_ylim((0, 2.6))\n    frame = data[data[\"frame\"] == frame_idx]\n    sns.scatterplot(frame, x=\"x\", y=\"y\", hue=\"type\", s=7, ax=ax)\n    ax.invert_yaxis()\n    ax.legend(loc='upper right')\n\ndata = data.drop(\"z\", axis=1)\nani = animation.FuncAnimation(\n    fig, animate_sequence, frames=sorted(data[\"frame\"].unique().tolist())[:])\nprint()","metadata":{"execution":{"iopub.status.busy":"2023-04-02T13:59:06.100361Z","iopub.execute_input":"2023-04-02T13:59:06.101026Z","iopub.status.idle":"2023-04-02T13:59:06.852227Z","shell.execute_reply.started":"2023-04-02T13:59:06.100975Z","shell.execute_reply":"2023-04-02T13:59:06.850764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# saving animation. Please, wait\n%time ani.save('myAnimation1.gif', writer='imagemagick', fps=30)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T13:59:12.866149Z","iopub.execute_input":"2023-04-02T13:59:12.866525Z","iopub.status.idle":"2023-04-02T14:00:14.143055Z","shell.execute_reply.started":"2023-04-02T13:59:12.866493Z","shell.execute_reply":"2023-04-02T14:00:14.141834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Animation ","metadata":{}},{"cell_type":"code","source":"HTML('<img src=\"./myAnimation1.gif\" />')","metadata":{"execution":{"iopub.status.busy":"2023-04-02T14:00:14.145486Z","iopub.execute_input":"2023-04-02T14:00:14.145854Z","iopub.status.idle":"2023-04-02T14:00:14.154066Z","shell.execute_reply.started":"2023-04-02T14:00:14.145816Z","shell.execute_reply":"2023-04-02T14:00:14.1528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# number of points on each frame\ngrouped_by_frames = data.groupby(\"frame\").count().reset_index()\nfig, ax = plt.subplots(1, 1, figsize=(10, 10))\nsns.histplot(grouped_by_frames, x=\"type\", bins=10)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T17:46:52.493435Z","iopub.execute_input":"2023-04-01T17:46:52.493887Z","iopub.status.idle":"2023-04-01T17:46:52.813774Z","shell.execute_reply.started":"2023-04-01T17:46:52.493849Z","shell.execute_reply":"2023-04-01T17:46:52.812872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filling nans, let it be ones\ndata.fillna(0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-01T17:53:09.332142Z","iopub.execute_input":"2023-04-01T17:53:09.332542Z","iopub.status.idle":"2023-04-01T17:53:09.34436Z","shell.execute_reply.started":"2023-04-01T17:53:09.332508Z","shell.execute_reply":"2023-04-01T17:53:09.342943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grouped_by_frames = data.groupby(\"frame\").count().reset_index()\ngrouped_by_frames","metadata":{"execution":{"iopub.status.busy":"2023-04-01T17:53:30.21951Z","iopub.execute_input":"2023-04-01T17:53:30.219957Z","iopub.status.idle":"2023-04-01T17:53:30.256386Z","shell.execute_reply.started":"2023-04-01T17:53:30.219919Z","shell.execute_reply":"2023-04-01T17:53:30.255177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict(tuple(data.groupby(\"frame\")))[103][[\"x\", \"y\", \"z\"]].to_numpy().flatten().shape","metadata":{"execution":{"iopub.status.busy":"2023-04-02T09:05:33.443315Z","iopub.execute_input":"2023-04-02T09:05:33.443822Z","iopub.status.idle":"2023-04-02T09:05:33.472242Z","shell.execute_reply.started":"2023-04-02T09:05:33.443774Z","shell.execute_reply":"2023-04-02T09:05:33.471191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Summary about one parquet file\n- It contains keypoints of face, hands and pose\n- the word is represented by the sequence of frames. On each frame human do some movements and to the end of all frames we have sequence of signs which compose a one word","metadata":{}},{"cell_type":"markdown","source":"# Things about model input format\nthere is three types of data representation in my mind:\n- (1) Take all coordinates of keypoints on single frame a flatten them. If one frame has D points than feature vector will be 3*D. Let N -- number of frames. Than we can get a sequence of length N where each item is the 3*D-dimensional vector. The shape of the one sample is (N, 3*D).\n- (2) Take a sequence of shape (N, 3*D) and calculate average of features (i think it's bad because of loosing temporal information but we can try this).\n\nI wrote dataloader and several models: lstm and transformer based models to fit on (1) variant of data representation and simple fully-connected neural network to fit on (2) variant of data representation. I tried to fit models but training process requires a lot of time. Basic free configuration of kaggle jupyter notebook does not allow fit models in background mode. I hope you will get some usefull ideas from this notebook (animation of data or data representation). \n\nI will be glad to discuss possible solutions","metadata":{}},{"cell_type":"code","source":"import torch\nfrom torch import nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom sklearn.model_selection import train_test_split\nimport json","metadata":{"execution":{"iopub.status.busy":"2023-04-02T09:10:12.052885Z","iopub.execute_input":"2023-04-02T09:10:12.053383Z","iopub.status.idle":"2023-04-02T09:10:12.059471Z","shell.execute_reply.started":"2023-04-02T09:10:12.053337Z","shell.execute_reply":"2023-04-02T09:10:12.05834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/asl-signs/train.csv\")\ntrain.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samples = train.sample(20000, random_state=42)\ntrain_anns, val_anns = train_test_split(samples, test_size=0.2, random_state=42, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T09:10:12.292075Z","iopub.execute_input":"2023-04-02T09:10:12.292548Z","iopub.status.idle":"2023-04-02T09:10:12.313303Z","shell.execute_reply.started":"2023-04-02T09:10:12.292504Z","shell.execute_reply":"2023-04-02T09:10:12.312311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class SignDataset(Dataset):\n    \n    def __init__(self, anns: pd.DataFrame, do_average=False):\n        self.anns = anns\n        self.do_average = do_average\n        with open(\"/kaggle/input/asl-signs/sign_to_prediction_index_map.json\") as f:\n            self.classes = json.load(f)\n        \n        \n    def __len__(self):\n        return len(self.anns)\n    \n    def __getitem__(self, idx):\n        \n        ann = self.anns.iloc[idx]\n        sequence = []\n        data = pq.read_table(\"/kaggle/input/asl-signs/\" + ann[\"path\"]).to_pandas().fillna(0)\n        frames = data[\"frame\"].unique().tolist()\n        grouped = dict(tuple(data.groupby(\"frame\")))\n        sequence = [torch.as_tensor(grouped[frame][[\"x\", \"y\", \"z\"]].to_numpy().flatten().astype(np.float32).reshape((1, 1629))) for frame in frames]\n\n        sequence = torch.cat(sequence)\n\n        if self.do_average:\n            sequence = sequence.mean(dim=0)\n        return sequence, self.classes[ann[\"sign\"]]\n    \ndef batch2xy(batch, device):\n    x = [sample[0] for sample in batch]\n    y = [sample[1] for sample in batch]\n    x = torch.nn.utils.rnn.pad_sequence(x, batch_first=True, padding_value=0).to(device)\n    y = torch.LongTensor(y).to(device)\n    \n    \n    return x, y","metadata":{"execution":{"iopub.status.busy":"2023-04-02T09:10:13.058574Z","iopub.execute_input":"2023-04-02T09:10:13.058949Z","iopub.status.idle":"2023-04-02T09:10:13.072036Z","shell.execute_reply.started":"2023-04-02T09:10:13.058919Z","shell.execute_reply":"2023-04-02T09:10:13.070901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Model(nn.Module):\n    \n    def __init__(self):\n        super().__init__()\n        self.lstm1 = nn.LSTM(1629, 1629//3, 20, batch_first=True)\n        self.lin1 = nn.Linear(1629//3, 500)\n        self.relu1 = nn.ReLU()\n        self.lin2 = nn.Linear(500, 250)\n        self.softmax = nn.Softmax(dim=1)\n        \n    def forward(self, x):\n        \n        x, hidden = self.lstm1(x)\n        x = x[:, -1]\n        x = self.relu1(self.lin1(x))\n        x = self.lin2(x)\n        x= self.softmax(x)\n        return x\n\nclass TransformerModel(nn.Module):\n    \n    def __init__(self):\n        super().__init__()\n        transformer_layer = nn.TransformerEncoderLayer(1629, nhead=9)\n        self.transofrmer_encoder = nn.TransformerEncoder(transformer_layer, 4)\n        self.lin1 = nn.Linear(1629, 250)\n        self.softmax = nn.Softmax(dim=1)\n        \n    def forward(self, x):\n        \n        x = self.transofrmer_encoder(x)\n#         print(x.shape)\n        x = x.mean(dim=1)\n        x = self.lin1(x)\n        x= self.softmax(x)\n        return x\n\nclass SimpleNN(nn.Module):\n    \n    def __init__(self):\n        super().__init__()\n        self.lin1 = nn.Linear(1629, 1629)\n        self.relu1 = nn.ReLU()\n        self.lin2 = nn.Linear(1629, 1629)\n        self.relu2 = nn.ReLU() \n        self.lin3 = nn.Linear(1629, 250)\n        self.softmax = nn.Softmax(dim=1)\n\n        \n    def forward(self, x):\n        \n        x = self.relu1(self.lin1(x))\n#         print(x.shape)\n        x = self.relu2(self.lin2(x))\n        x = self.lin3(x)\n        x= self.softmax(x)\n        return x\n    ","metadata":{"execution":{"iopub.status.busy":"2023-04-02T09:10:13.901326Z","iopub.execute_input":"2023-04-02T09:10:13.902341Z","iopub.status.idle":"2023-04-02T09:10:13.914875Z","shell.execute_reply.started":"2023-04-02T09:10:13.902293Z","shell.execute_reply":"2023-04-02T09:10:13.91361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train loop","metadata":{}},{"cell_type":"code","source":"epochs = 100\nbatch_size = 32\n\ntrain_dataset = SignDataset(train_anns, do_average=True)\nval_dataset = SignDataset(val_anns, do_average=True)\nprint(\"train size: \", len(train_dataset))\nprint(\"val size: \", len(val_dataset))\n\ntrain_loader = DataLoader(train_dataset, batch_size=batch_size, num_workers=2, collate_fn=lambda x: x)\nval_loader = DataLoader(val_dataset, batch_size=batch_size, num_workers=2, collate_fn=lambda x: x)\n\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nprint(device)\nmodel = SimpleNN().to(device)\nprint(\"model parameters: \", sum(p.numel() for p in model.parameters()))\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.RAdam(model.parameters(), lr=0.01)","metadata":{"execution":{"iopub.status.busy":"2023-04-02T09:11:14.196224Z","iopub.execute_input":"2023-04-02T09:11:14.197154Z","iopub.status.idle":"2023-04-02T09:11:14.27076Z","shell.execute_reply.started":"2023-04-02T09:11:14.197103Z","shell.execute_reply":"2023-04-02T09:11:14.269712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loss_list = []\nval_loss_list = []\ncntr = 0\nfor epoch in range(epochs):\n    \n    #train\n    print(\"epoch: \", epoch)\n    train_loss = 0\n    val_loss = 0\n    num_batches = 0\n    model.train()\n    for batch in train_loader:\n        cntr += 1\n#         print(\"iter: \", cntr)\n        optimizer.zero_grad()\n        x, y = batch2xy(batch, device)\n        out = model(x)\n        loss = criterion(out, y)\n        loss.backward()\n        optimizer.step()\n        train_loss += loss.item()\n        num_batches += 1\n        if cntr % 200 == 0:\n            print(f\"iter: {cntr}, train loss: \", train_loss/num_batches)\n        \n    train_loss_list.append(train_loss/num_batches)\n        \n    print(\"train: \", train_loss_list[-1])\n    with torch.no_grad():\n        num_batches = 0\n        model.eval()\n        for sample in val_loader:\n\n            x, y = batch2xy(batch, device)\n            out = model(x)\n            loss = criterion(out, y)\n            val_loss += loss.item()\n            num_batches += 1\n        val_loss_list.append(val_loss/num_batches)\n        print(\"val: \", val_loss_list[-1])","metadata":{"execution":{"iopub.status.busy":"2023-04-02T09:11:14.428904Z","iopub.execute_input":"2023-04-02T09:11:14.429289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"val: \", val_loss_list[-1])","metadata":{"execution":{"iopub.status.busy":"2023-04-01T22:36:43.559747Z","iopub.execute_input":"2023-04-01T22:36:43.560579Z","iopub.status.idle":"2023-04-01T22:36:43.573697Z","shell.execute_reply.started":"2023-04-01T22:36:43.560531Z","shell.execute_reply":"2023-04-01T22:36:43.572511Z"},"trusted":true},"execution_count":null,"outputs":[]}]}