{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport json\nfrom tqdm import tqdm\nimport numpy as np\nimport pandas as pd\n\nimport torch\nimport torch.nn as nn\nimport multiprocessing as mp","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-28T19:06:32.810820Z","iopub.execute_input":"2023-02-28T19:06:32.811189Z","iopub.status.idle":"2023-02-28T19:06:32.815937Z","shell.execute_reply.started":"2023-02-28T19:06:32.811161Z","shell.execute_reply":"2023-02-28T19:06:32.815053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LANDMARK_FILES_DIR = \"/kaggle/input/asl-signs/train_landmark_files\"\nTRAIN_FILE = \"/kaggle/input/asl-signs/train.csv\"\nlabel_map = json.load(open(\"/kaggle/input/asl-signs/sign_to_prediction_index_map.json\", \"r\"))","metadata":{"execution":{"iopub.status.busy":"2023-02-28T19:06:09.502239Z","iopub.execute_input":"2023-02-28T19:06:09.502770Z","iopub.status.idle":"2023-02-28T19:06:09.510928Z","shell.execute_reply.started":"2023-02-28T19:06:09.502733Z","shell.execute_reply":"2023-02-28T19:06:09.509633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class FeatureGen(nn.Module):\n    def __init__(self):\n        super(FeatureGen, self).__init__()\n        pass\n    \n    def forward(self, x):\n        face_x = x[:,:468,:].contiguous().view(-1, 468*3)\n        lefth_x = x[:,468:489,:].contiguous().view(-1, 21*3)\n        pose_x = x[:,489:522,:].contiguous().view(-1, 33*3)\n        righth_x = x[:,522:,:].contiguous().view(-1, 21*3)\n        \n        lefth_x = lefth_x[~torch.any(torch.isnan(lefth_x), dim=1),:]\n        righth_x = righth_x[~torch.any(torch.isnan(righth_x), dim=1),:]\n        \n        x1m = torch.mean(face_x, 0)\n        x2m = torch.mean(lefth_x, 0)\n        x3m = torch.mean(pose_x, 0)\n        x4m = torch.mean(righth_x, 0)\n        \n        x1s = torch.std(face_x, 0)\n        x2s = torch.std(lefth_x, 0)\n        x3s = torch.std(pose_x, 0)\n        x4s = torch.std(righth_x, 0)\n        \n        xfeat = torch.cat([x1m,x2m,x3m,x4m, x1s,x2s,x3s,x4s], axis=0)\n        xfeat = torch.where(torch.isnan(xfeat), torch.tensor(0.0, dtype=torch.float32), xfeat)\n        \n        return xfeat\n    \nfeature_converter = FeatureGen()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T19:06:09.512274Z","iopub.execute_input":"2023-02-28T19:06:09.512590Z","iopub.status.idle":"2023-02-28T19:06:09.523876Z","shell.execute_reply.started":"2023-02-28T19:06:09.512551Z","shell.execute_reply":"2023-02-28T19:06:09.522963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ROWS_PER_FRAME = 543\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-02-28T19:06:09.525410Z","iopub.execute_input":"2023-02-28T19:06:09.525895Z","iopub.status.idle":"2023-02-28T19:06:09.538869Z","shell.execute_reply.started":"2023-02-28T19:06:09.525868Z","shell.execute_reply":"2023-02-28T19:06:09.537764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convert_row(row):\n    x = load_relevant_data_subset(os.path.join(\"/kaggle/input/asl-signs\", row[1].path))\n    x = feature_converter(torch.tensor(x)).cpu().numpy()\n    return x, row[1].label\n\ndef convert_and_save_data():\n    df = pd.read_csv(TRAIN_FILE)\n    df['label'] = df['sign'].map(label_map)\n    npdata = np.zeros((df.shape[0], 3258))\n    nplabels = np.zeros(df.shape[0])\n    with mp.Pool() as pool:\n        results = pool.imap(convert_row, df.iterrows(), chunksize=250)\n        for i, (x,y) in tqdm(enumerate(results), total=df.shape[0]):\n            npdata[i,:] = x\n            nplabels[i] = y\n    \n    np.save(\"feature_data.npy\", npdata)\n    np.save(\"feature_labels.npy\", nplabels)\n        \nconvert_and_save_data()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T19:06:36.934707Z","iopub.execute_input":"2023-02-28T19:06:36.935119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}