{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import parckages","metadata":{}},{"cell_type":"code","source":"!pip install pyarrow\n!pip install seaborn","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:13:52.243352Z","iopub.execute_input":"2023-06-18T15:13:52.243779Z","iopub.status.idle":"2023-06-18T15:14:01.872914Z","shell.execute_reply.started":"2023-06-18T15:13:52.243747Z","shell.execute_reply":"2023-06-18T15:14:01.871739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Numerical Operations\nimport numpy as np\nimport math\n\n# Reading/Writing Data\nimport pandas as pd\nimport os\nimport sys\nimport json\n\n# Memory menagement\nimport gc\n\n# Plotting \nimport matplotlib.pyplot as plt\nimport seaborn as sns\n# import plotly.express as px\n\n# For Progress Bar\n# from tqdm.notebook import tqdm\nfrom tqdm import tqdm\n\n# Pytorch\nimport torch\nfrom torch.utils.data import Dataset, DataLoader, random_split\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torchaudio\n\n# Tensorflow\nfrom tensorflow.keras.utils import to_categorical\nimport tensorflow as tf\n\n# Processing data\nfrom multiprocessing import Pool\nfrom scipy.ndimage import zoom\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import train_test_split\n\n# For plotting learning curve\nfrom torch.utils.tensorboard import SummaryWriter\n\n# set the maximum number of rows to display to 500\npd.set_option('display.max_rows', 500)\n\nprint(f'Python V{sys.version}')","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:14:01.874965Z","iopub.execute_input":"2023-06-18T15:14:01.875359Z","iopub.status.idle":"2023-06-18T15:14:01.886625Z","shell.execute_reply.started":"2023-06-18T15:14:01.875325Z","shell.execute_reply":"2023-06-18T15:14:01.885644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utility functions","metadata":{}},{"cell_type":"code","source":"ENV = ''","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:14:01.887757Z","iopub.execute_input":"2023-06-18T15:14:01.888065Z","iopub.status.idle":"2023-06-18T15:14:01.899426Z","shell.execute_reply.started":"2023-06-18T15:14:01.888039Z","shell.execute_reply":"2023-06-18T15:14:01.898546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def same_seed(seed): \n    '''Fixes random number generator seeds for reproducibility.'''\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    if torch.cuda.is_available():\n        torch.cuda.manual_seed_all(seed)\n\n# Landmark\n# ---\n# lips points\n# https://www.kaggle.com/competitions/asl-signs/discussion/391812#2168354\n# lipsUpperOuter =  [61, 185, 40, 39, 37, 0, 267, 269, 270, 409, 291]\n# lipsLowerOuter = [146, 91, 181, 84, 17, 314, 405, 321, 375, 291]\n# lipsUpperInner = [78, 191, 80, 81, 82, 13, 312, 311, 310, 415, 308]\n# lipsLowerInner = [78, 95, 88, 178, 87, 14, 317, 402, 318, 324, 308]\n# lips = lipsUpperOuter + lipsLowerOuter + lipsUpperInner + lipsLowerInner\nLIP_IDXS = [191, 80, 81, 82, 13, 312, 311, 310, 415, 95, \n            88, 178, 87, 14, 317, 402, 318, 324]\n\n# left_hand: 468:489\n\n# pose points\n# 0-24 -> 489:513\n\n# right_hand: 522:543\n# ---\n\ndef get_full_path(path):\n    return os.path.join(\"/kaggle/input/asl-signs/\", path)\n\n# For sign to index convertion\nSIGN_DICT = {}\nwith open('/kaggle/input/asl-signs/sign_to_prediction_index_map.json', \"r\") as f:\n    SIGN_DICT = json.load(f)\n\ndef create_features(row):\n    path = row[1].path\n    sign = row[1].sign\n    df = pd.read_parquet(get_full_path(path), columns=['x', 'y', 'z'])\n    n_frames = int(df.shape[0]/543)\n    features = df.values.reshape(n_frames, 543, 3)\n\n    # select points\n    lip = features[:, LIP_IDXS, :]\n    left_hand = features[:, 468:489, :]\n    pose = features[:, 489:513, :]\n    right_hand = features[:, 522:543, :]\n    # (n_frame, 84, 3)\n    data = np.concatenate((lip, left_hand, pose, right_hand), axis=1)\n\n    # fillna\n    data = np.nan_to_num(data, nan=0)\n    \n    # interpolate ('nearest' mode used in F.interpolate of PyTorch)\n    desired_shape = (64, 84, 3)\n    data = zoom(data, (desired_shape[0] / data.shape[0], 1, 1), order=0)\n    \n    # For CNN (3, n_frame, 84)\n    data = np.transpose(data, (2, 0, 1))\n    \n    # concat 3D coordinates\n#     data = data.reshape((data.shape[0], 84*3))\n    \n    return data, SIGN_DICT.get(sign)\n\ndef preprocess_data(train_path, seed):\n    # Features\n    if ENV == \"DEV\":\n        train_df = pd.read_csv(train_path).sample(int(5e2), random_state=seed)\n#         train_df = pd.read_csv(train_path).sample(int(4e4), random_state=seed)\n    \n    else:\n        train_df = pd.read_csv(train_path).sample(frac=1, random_state=seed) # suffle\n        \n    # Print out shape\n    n_samples = len(train_df)\n    print(f'N_SAMPLES: {n_samples}')\n    \n    # Multiprocessing to speed up\n    with Pool() as pool:\n        results = list(tqdm(pool.imap(create_features, train_df.iterrows(), chunksize=250), total=len(train_df)))\n\n    data_X = np.array([res[0] for res in results])\n    data_y = np.array([res[1] for res in results])\n\n    # Flatten for NN\n#     data_X = data_X.reshape(data_X.shape[0], -1)\n    \n    print(data_X.shape)\n    print(data_y.shape)\n    \n    return data_X, data_y\n\ndef train_valid_split(data_X, data_y, val_ratio):\n    train_set_size = int(len(data_X) * (1 - val_ratio))\n    train_X, val_X = data_X[:train_set_size], data_X[train_set_size:]\n    train_y, val_y = data_y[:train_set_size], data_y[train_set_size:]\n    \n    return np.array(train_X), np.array(train_y), np.array(val_X), np.array(val_y)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:14:01.901762Z","iopub.execute_input":"2023-06-18T15:14:01.902075Z","iopub.status.idle":"2023-06-18T15:14:01.928682Z","shell.execute_reply.started":"2023-06-18T15:14:01.902048Z","shell.execute_reply":"2023-06-18T15:14:01.927691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"class ASLDataset(Dataset):\n    '''\n    x: Features.\n    y: Targets, if none, do prediction.\n    '''\n    def __init__(self, x, y=None):\n        if y is None:\n            self.y = y\n        else:\n            self.y = torch.LongTensor(y)\n        self.x = torch.FloatTensor(x)\n        \n        # TODO: using pipeline\n#         self.spec_aug = torch.nn.Sequential(\n#                 torchaudio.transforms.FrequencyMasking(80),\n#                 torchaudio.transforms.TimeMasking(18),\n#             )\n\n    def __getitem__(self, idx):\n        if self.y is None:\n            return self.x[idx]\n        else:\n            return self.x[idx], self.y[idx]\n\n    def __len__(self):\n        return len(self.x)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:14:01.929796Z","iopub.execute_input":"2023-06-18T15:14:01.930072Z","iopub.status.idle":"2023-06-18T15:14:01.946071Z","shell.execute_reply.started":"2023-06-18T15:14:01.930049Z","shell.execute_reply":"2023-06-18T15:14:01.945219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"device = 'cuda' if torch.cuda.is_available() else 'cpu'\n\nSEED = 2023  # seed number\n\n# Dataset\nVALID_RATIO = 0.2\nBATCH_SIZE = 256\n\n# Model parameters\n# INPUT_DIM from train_X.shape[2]\nHIDDEN_LAYERS = 4                   # the number of hidden layers\nHIDDEN_DIM = 1024                   # the hidden dim\nOUTPUT_DIM = 250                    # Number of class\n\n# Training parameters\nN_EPOCHS = 100\nLEARNING_RATE = 1e-3\nWEIGHT_DECAY = 1e-6\nEARLY_STOP = 300\n\n# Save\nSAVE_PATH = './models/model.ckpt'  # Your model will be saved here.\nENABLE_TENSORBOARD = False","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:14:01.947117Z","iopub.execute_input":"2023-06-18T15:14:01.947426Z","iopub.status.idle":"2023-06-18T15:14:01.960843Z","shell.execute_reply.started":"2023-06-18T15:14:01.947401Z","shell.execute_reply":"2023-06-18T15:14:01.960077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataloader","metadata":{}},{"cell_type":"code","source":"# Set seed for reproducibility\nsame_seed(SEED)\n\n# preprocess data\ndata_X, data_y = preprocess_data('/kaggle/input/asl-signs/train.csv', SEED)\ntrain_X, train_y, val_X, val_y = train_valid_split(data_X, data_y, val_ratio=VALID_RATIO)\n# LSTM input dim is (batch_size, seq_len, input_size) seq_len can ignore\nINPUT_DIM = train_X.shape[1]\n\n# remove raw feature to save memory\ndel data_X, data_y\ngc.collect()\n\n# get dataset\ntrain_dataset, val_dataset = ASLDataset(train_X, train_y), \\\n                             ASLDataset(val_X, val_y)\n\n# remove raw feature to save memory\ndel train_X, train_y, val_X, val_y\ngc.collect()\n\n# get dataloader\ntrain_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=BATCH_SIZE, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:14:01.961861Z","iopub.execute_input":"2023-06-18T15:14:01.962140Z","iopub.status.idle":"2023-06-18T15:15:07.779715Z","shell.execute_reply.started":"2023-06-18T15:14:01.962115Z","shell.execute_reply":"2023-06-18T15:15:07.778732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model","metadata":{}},{"cell_type":"code","source":"# --- LSTM ---\nclass LSTMBlock(nn.Module):\n    def __init__(self, input_dim, output_dim):\n        super(LSTMBlock, self).__init__()\n        \n        self.block = nn.Sequential(\n            nn.LSTM(input_dim, output_dim, batch_first=True)\n        )\n        \n    def forward(self, x):\n        x, _ = self.block(x)\n        return x\n\nclass LSTMClassifier(nn.Module):\n    def __init__(self, input_dim, hidden_dim=32, hidden_layers=1, output_dim=250):\n        super(Classifier, self).__init__()\n        \n        self.layers = nn.Sequential(\n            LSTMBlock(input_dim, hidden_dim),\n            *[LSTMBlock(hidden_dim, hidden_dim) for _ in range(hidden_layers)],\n        )\n        self.fc = nn.Linear(hidden_dim, output_dim)\n        \n    def forward(self, x):\n        x = self.layers(x)\n        x = self.fc(x[:, -1, :])\n        return x \n# ------\n\n# --- NN ---\nclass NNBlock(nn.Module):\n    def __init__(self, input_dim, output_dim):\n        super(NNBlock, self).__init__()\n        \n        self.block = nn.Sequential(\n            nn.Linear(input_dim, output_dim)\n        )\n        \n    def forward(self, x):\n        x = self.block(x)\n        return x\n\n    \nclass NNClassifier(nn.Module):\n    def __init__(self, input_dim, hidden_dim=32, hidden_layers=1, output_dim=250):\n        super(Classifier, self).__init__()\n        \n        self.layers = nn.Sequential(\n            NNBlock(input_dim, hidden_dim),\n            *[NNBlock(hidden_dim, hidden_dim) for _ in range(hidden_layers)],\n        )\n        self.fc = nn.Linear(hidden_dim, output_dim)\n        \n    def forward(self, x):\n        x = self.layers(x)\n        return x \n# ------\n\n# --- Transformer ---\nclass TransformerBlock(nn.Module):\n    def __init__(self, input_dim, output_dim):\n        super(TransformerBlock, self).__init__()\n        \n        self.block = nn.Sequential(\n            nn.Linear(input_dim, output_dim)\n        )\n        \n    def forward(self, x):\n        x = self.block(x)\n        return x\n\n    \nclass TransformerClassifier(nn.Module):\n    def __init__(self, input_dim, hidden_dim=32, hidden_layers=1, output_dim=250):\n        super(Classifier, self).__init__()\n        \n        self.layers = nn.Sequential(\n            TransformerBlock(input_dim, hidden_dim),\n            *[TransformerBlock(hidden_dim, hidden_dim) for _ in range(hidden_layers)],\n        )\n        self.fc = nn.Linear(hidden_dim, output_dim)\n        \n    def forward(self, x):\n        x = self.layers(x)\n        return x \n\nclass Classifier(nn.Module):\n    def __init__(self, input_dim, d_model=256, n_class=250, dropout=0.3):\n        super().__init__()\n        # Project the dimension of features from that of input into d_model.\n        self.prenet = nn.Linear(input_dim, d_model)\n        self.encoder_layer = nn.TransformerEncoderLayer(d_model=d_model, dim_feedforward=512, nhead=4)\n        self.encoder = nn.TransformerEncoder(self.encoder_layer, num_layers=2)\n        \n        # Project the the dimension of features from d_model into speaker nums.\n        self.pred_layer = nn.Sequential(\n            nn.Linear(d_model, d_model),\n            nn.ReLU(),\n            nn.Linear(d_model, n_spks),\n        )\n        \n    def forward(self, mels):\n        \"\"\"\n        args:\n            mels: (batch size, length, 40)\n        return:\n            out: (batch size, n_spks)\n        \"\"\"\n        # out: (batch size, length, d_model)\n        out = self.prenet(mels)\n        # out: (length, batch size, d_model)\n        out = out.permute(1, 0, 2)\n        # The encoder layer expect features in the shape of (length, batch size, d_model).\n        out = self.encoder(out)\n        # out: (batch size, length, d_model)\n        out = out.transpose(0, 1)\n        # mean pooling\n        stats = out.mean(dim=1)\n\n        # out: (batch, n_spks)\n        out = self.pred_layer(stats)\n        return out\n# ------\n\n# --- CNN ---\nclass CNNClassifier(nn.Module):\n    def __init__(self):\n        super(CNNClassifier, self).__init__()\n        self.cnn = nn.Sequential(\n            nn.BatchNorm2d(3, eps=1e-5),\n            nn.Conv2d(3,\n                      3,\n                      kernel_size=(3, 3),\n                      padding=1),\n#             nn.AdaptiveAvgPool2d((1, 1)),\n            nn.Flatten(),\n            nn.Dropout(0.5),\n            nn.Linear(64*84*3, 250)\n        )\n\n    def forward(self, x):\n        out = self.cnn(x)\n        return out\n# ------","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:15:07.781610Z","iopub.execute_input":"2023-06-18T15:15:07.781959Z","iopub.status.idle":"2023-06-18T15:15:07.810469Z","shell.execute_reply.started":"2023-06-18T15:15:07.781929Z","shell.execute_reply":"2023-06-18T15:15:07.809475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def model_summary(model):\n#     for layer in model.children():\n#         print(\"Layer : {}\".format(layer))\n#         print(\"Parameters : \")\n#         for param in layer.parameters():\n#             print(param.shape)\n#         print()\n        \n# model = Classifier(input_dim=84*3, hidden_layers=HIDDEN_LAYERS, hidden_dim=HIDDEN_DIM).to(device)\n# model_summary(model)","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:15:07.812072Z","iopub.execute_input":"2023-06-18T15:15:07.812488Z","iopub.status.idle":"2023-06-18T15:15:07.826322Z","shell.execute_reply.started":"2023-06-18T15:15:07.812461Z","shell.execute_reply":"2023-06-18T15:15:07.825493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training","metadata":{}},{"cell_type":"code","source":"# for ploting loss each epoch\ntrain_loss_hist = np.zeros(N_EPOCHS)\nvalid_loss_hist = np.zeros(N_EPOCHS)\n\nwriter = SummaryWriter() # Writer of tensoboard.\n\n# create model, define a loss function, and optimizer\n# model = Classifier(input_dim=INPUT_DIM, hidden_layers=HIDDEN_LAYERS, hidden_dim=HIDDEN_DIM).to(device)\nmodel = CNNClassifier().to(device)\ncriterion = nn.CrossEntropyLoss()\noptimizer = torch.optim.Adam(model.parameters(), lr=LEARNING_RATE, weight_decay=WEIGHT_DECAY)\n# sched = torch.optim.lr_scheduler.StepLR(opt, step_size=100, gamma=0.95)\n\nif not os.path.isdir('./models'):\n    os.mkdir('./models') # Create directory of saving models.\n\nbest_acc, step, early_stop_count = 0, 0, 0\n\nfor epoch in range(N_EPOCHS):\n    train_acc, train_loss, val_acc, val_loss = 0.0, 0.0, 0.0, 0.0\n    \n    # Training\n    train_pbar = tqdm(train_loader, position=0, leave=True)  # visualize training progress.\n    model.train()                           # Set your model to train mode.\n    for x, y in train_pbar:\n        optimizer.zero_grad()               # Set gradient to zero.\n        x, y = x.to(device), y.to(device)   # Move your data to device. \n        output = model(x)\n        loss = criterion(output, y)\n        loss.backward()                     # Compute gradient(backpropagation).\n        optimizer.step()                    # Update parameters.\n        step += 1                           # For Tensorbord\n\n        _, train_pred = torch.max(output, 1) # get the index of the class with the highest probability\n        train_acc += (train_pred.detach() == y.detach()).sum().item()\n        train_loss += loss.detach().item()\n\n        # Display current epoch number and loss on tqdm progress bar.\n        train_pbar.set_description(f'Epoch [{epoch+1}/{N_EPOCHS}]')\n        train_pbar.set_postfix({'loss': loss.detach().item()})\n\n    # validation\n    val_pbar = tqdm(val_loader, position=0, leave=True)\n    model.eval()  # Set your model to evaluation mode.\n    for x, y in val_pbar:\n        x, y = x.to(device), y.to(device)\n        with torch.no_grad():\n            output = model(x)\n            loss = criterion(output, y)\n\n        _, val_pred = torch.max(output, 1) \n        val_acc += (val_pred.cpu() == y.cpu()).sum().item()\n        val_loss += loss.detach().item()\n\n        # Display current epoch number and loss on tqdm progress bar.\n        val_pbar.set_description(f'Epoch [{epoch+1}/{N_EPOCHS}]')\n        val_pbar.set_postfix({'loss': loss.detach().item()})\n    \n    # Acc/Loss\n    mean_train_loss = train_loss/len(train_loader)\n    mean_valid_loss = val_loss/len(val_loader)\n    print(f'Train Acc: {train_acc/len(train_dataset):3.5f} Loss: {mean_train_loss:3.5f} | Val Acc: {val_acc/len(val_dataset):3.5f} loss: {mean_valid_loss:3.5f}')\n    \n    # Record loss\n    train_loss_hist[epoch] = mean_train_loss\n    valid_loss_hist[epoch] = mean_valid_loss\n    \n    # Record data for Tensorbord\n    if ENABLE_TENSORBOARD:\n        writer.add_scalar('Loss/train', train_loss/len(train_loader), step)\n        writer.add_scalar('Loss/valid', val_loss/len(valid_loader), step)\n\n    # Checkpoint, save best model\n    if val_acc > best_acc:\n        best_acc = val_acc\n        torch.save(model.state_dict(), SAVE_PATH) # Save your best model\n        print('Saving model with acc {:.5f}...'.format(best_acc/len(val_dataset)))\n        early_stop_count = 0\n    else:\n        early_stop_count += 1\n\n    # Early stop\n    if early_stop_count >= EARLY_STOP:\n        print('\\nModel is not improving, so we halt the training session.')","metadata":{"execution":{"iopub.status.busy":"2023-06-18T15:15:07.829443Z","iopub.execute_input":"2023-06-18T15:15:07.829701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create the line plot using Seaborn\nsns.lineplot(train_loss_hist, label='train')\nsns.lineplot(valid_loss_hist, label='val')\n\n# Set plot labels and title\nplt.xlabel('epoch')\nplt.ylabel('loss')\nplt.title('Loss')\n\n# Display the legend\nplt.legend()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del train_dataset, val_dataset\n# del train_loader, val_loader\n# gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!nvcc --version","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ENV = \"DEV\"\n\n# SEED = 42\n\n# if ENV == \"DEV\":\n#     train_df = pd.read_csv('/kaggle/input/asl-signs/train.csv').sample(int(5e3), random_state=SEED)\n# else:\n#     train_df = pd.read_csv('/kaggle/input/asl-signs/train.csv')\n    \n# with open(\"/kaggle/input/asl-signs/sign_to_prediction_index_map.json\", \"r\") as f:\n#     dict = json.load(f)\n\n# N_SAMPLES = len(train_df)\n# print(f'N_SAMPLES: {N_SAMPLES}')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def get_full_path(path):\n#     return os.path.join(\"/kaggle/input/asl-signs/\", path)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# interpolate","metadata":{}},{"cell_type":"code","source":"# lips points\n# https://www.kaggle.com/competitions/asl-signs/discussion/391812#2168354\n# lipsUpperOuter =  [61, 185, 40, 39, 37, 0, 267, 269, 270, 409, 291]\n# lipsLowerOuter = [146, 91, 181, 84, 17, 314, 405, 321, 375, 291]\n# lipsUpperInner = [78, 191, 80, 81, 82, 13, 312, 311, 310, 415, 308]\n# lipsLowerInner = [78, 95, 88, 178, 87, 14, 317, 402, 318, 324, 308]\n# lips = lipsUpperOuter + lipsLowerOuter + lipsUpperInner + lipsLowerInner\nLIP_IDXS = [191, 80, 81, 82, 13, 312, 311, 310, 415, 95, \n            88, 178, 87, 14, 317, 402, 318, 324]\n\n# left_hand: 468:489\n\n# pose points\n# 0-24 -> 489:513\n\n# right_hand: 522:543","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def create_features(row):\n#     path = row[1].path\n#     sign = row[1].sign\n#     df = pd.read_parquet(get_full_path(path), columns=['x', 'y', 'z'])\n#     n_frames = int(df.shape[0]/543)\n#     features = df.values.reshape(n_frames, 543, 3)\n\n#     # select points\n#     lip = features[:, LIP_IDXS, :]\n#     left_hand = features[:, 468:489, :]\n#     pose = features[:, 489:513, :]\n#     right_hand = features[:, 522:543, :]\n#     # (n_frame, 85, 3)\n#     data = np.concatenate((lip, left_hand, pose, right_hand), axis=1)\n\n#     # standard normalization\n# #     mean = np.nanmean(data)\n# #     std = np.nanstd(data)\n# #     data = (data - mean) / std\n\n#     # fillna\n#     data = np.nan_to_num(data, nan=0)\n    \n#     # interpolate ('nearest' mode used in F.interpolate of PyTorch)\n#     desired_shape = (64, 84, 3)\n#     data = zoom(data, (desired_shape[0] / data.shape[0], 1, 1), order=0)\n    \n#     data = data.reshape((data.shape[0], 84*3))\n    \n#     return data, dict.get(sign)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with Pool() as pool:\n#     results = list(tqdm(pool.imap(create_features, train_df.iterrows(), chunksize=250), total=len(train_df)))\n    \n# data_X = np.array([res[0] for res in results])\n# data_y = np.array([res[1] for res in results])\n\n# print(data_X.shape)\n# print(data_y.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LSTM","metadata":{}},{"cell_type":"code","source":"# class ASLData(Dataset):\n#     def __init__(self, datax, datay):\n#         self.datax = datax\n#         self.datay = datay\n        \n#     def __getitem__(self, index):\n#         return self.datax[index,:], self.datay[index]\n        \n#     def __len__(self):\n#         return len(self.datay)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class LSTMModel(nn.Module):\n#     def __init__(self):\n#         super(LSTMModel, self).__init__()\n#         self.lstm1 = nn.LSTM(84*3, 64, batch_first=True)\n#         self.lstm2 = nn.LSTM(64, 128, batch_first=True)\n#         self.lstm3 = nn.LSTM(128, 32, batch_first=True)\n#         self.fc1 = nn.Linear(64*32, 1024)\n#         self.fc2 = nn.Linear(1024, 512)\n#         self.fc3 = nn.Linear(512, 250)\n#         self.relu = nn.ReLU()\n        \n#     def forward(self, x):\n#         out, _ = self.lstm1(x)\n#         out, _ = self.lstm2(out)\n#         out, _ = self.lstm3(out)\n#         out = out.reshape(out.shape[0], -1)\n#         out = self.fc1(out)\n#         out = F.relu(out)\n#         out = self.fc2(out)\n#         out = F.relu(out)\n#         out = self.fc3(out)\n#         return out","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# EPOCHS = 1000\n# BATCH_SIZE = 64\n# N_SPLITS = 5  # Number of cross-validation folds\n\n# train_loss = np.zeros((N_SPLITS, EPOCHS))\n# train_acc = np.zeros((N_SPLITS, EPOCHS))\n# val_loss = np.zeros((N_SPLITS, EPOCHS))\n# val_acc = np.zeros((N_SPLITS, EPOCHS))\n\n# kf = KFold(n_splits=N_SPLITS, shuffle=True, random_state=42)\n\n# # Iterate over cross-validation folds\n# fold = 0\n# for train_idx, val_idx in kf.split(data_X):\n#     print(f\"\\n\\n... Training on Fold: {fold+1}\\n\")\n#     trainx, valx = data_X[train_idx], data_X[val_idx]\n#     trainy, valy = data_y[train_idx], data_y[val_idx]\n\n#     train_data = ASLData(trainx, trainy)\n#     val_data = ASLData(valx, valy)\n\n#     train_loader = DataLoader(train_data, batch_size=BATCH_SIZE, num_workers=4, shuffle=True)\n#     val_loader = DataLoader(val_data, batch_size=BATCH_SIZE, num_workers=4, shuffle=False)\n\n#     model = LSTMModel()\n#     opt = torch.optim.Adam(model.parameters(), lr=0.005)\n#     criterion = nn.CrossEntropyLoss()\n#     sched = torch.optim.lr_scheduler.StepLR(opt, step_size=300, gamma=0.95)\n\n#     for epoch in range(EPOCHS):\n#         model.train()\n#         train_loss_sum = 0.\n#         train_correct = 0\n#         train_total = 0\n\n#         for x, y in train_loader:\n#             x = torch.Tensor(x).float()\n#             y = torch.Tensor(y).long()\n\n#             y_pred = model(x)\n\n#             loss = criterion(y_pred, y)\n#             loss.backward()\n#             opt.step()\n#             opt.zero_grad()\n\n#             train_loss_sum += loss.item()\n#             train_correct += np.sum((np.argmax(y_pred.detach().cpu().numpy(), axis=1) == y.cpu().numpy()))\n#             train_total += 1\n#             sched.step()\n\n#         val_loss_sum = 0.\n#         val_correct = 0\n#         val_total = 0\n\n#         model.eval()\n#         for x, y in val_loader:\n#             x = torch.Tensor(x).float()\n#             y = torch.Tensor(y).long()\n\n#             with torch.no_grad():\n#                 y_pred = model(x)\n#                 loss = criterion(y_pred, y)\n#                 val_loss_sum += loss.item()\n#                 val_correct += np.sum((np.argmax(y_pred.cpu().numpy(), axis=1) == y.cpu().numpy()))\n#                 val_total += 1\n\n#         train_loss[fold, epoch] = train_loss_sum / train_total\n#         train_acc[fold, epoch] = train_correct / len(train_data)\n#         val_loss[fold, epoch] = val_loss_sum / val_total\n#         val_acc[fold, epoch] = val_correct / len(val_data)\n\n#         print(f\"Epoch:{epoch} > Train Loss: {(train_loss_sum / train_total):.04f}, Train Acc: {train_correct / len(train_data):0.04f}\")\n#         print(f\"Epoch:{epoch} > Val Loss: {(val_loss_sum / val_total):.04f}, Val Acc: {val_correct / len(val_data):0.04f}\")\n#         print(\"=\" * 50)\n\n#     fold += 1","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tensorflow ver.","metadata":{}},{"cell_type":"code","source":"# from tensorflow.keras.models import Sequential\n# from tensorflow.keras.layers import LSTM, Dense\n# from tensorflow.keras.callbacks import TensorBoard\n\n# y = to_categorical(data_y).astype(int)\n# X_train, X_test, y_train, y_test = train_test_split(data_X, y, test_size=0.1)\n\n# model = Sequential()\n# model.add(LSTM(64, return_sequences=True, activation='relu', input_shape=(64,252)))\n# model.add(LSTM(128, return_sequences=True, activation='relu'))\n# model.add(LSTM(64, return_sequences=False, activation='relu'))\n# model.add(Dense(64, activation='relu'))\n# model.add(Dense(32, activation='relu'))\n# model.add(Dense(250, activation='softmax'))\n\n# model.compile(optimizer='Adam', loss='categorical_crossentropy', metrics=['categorical_accuracy'])\n# model.fit(X_train, y_train, epochs=2000)\n# model.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tmp.shape)\n\nimages_transposed = np.transpose(tmp, (2, 0, 1))\n\nprint(images_transposed.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\n# Create a figure with three subplots\nfig, axes = plt.subplots(1, 3)\n\n# Plot the first image\naxes[0].imshow(tmp[:,:,0], cmap='summer', interpolation='nearest')\naxes[0].set_title('x')\n\n# Plot the second image\naxes[1].imshow(tmp[:,:,1], cmap='summer', interpolation='nearest')\naxes[1].set_title('y')\n\n# Plot the third image\naxes[2].imshow(tmp[:,:,2], cmap='summer', interpolation='nearest')\naxes[2].set_title('z')\n\n# Adjust spacing between subplots\nplt.tight_layout()\n\n# Add colorbar\n# plt.colorbar()\n\n# Show the plot\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}