{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n'''\nFunction selecting train and test parquets from the train.csv\n'''\ndef select_participants(df, selected_signs, seed, train_size, test_size):\n    df_train = pd.DataFrame()\n    df_test = pd.DataFrame()\n    # Select rows with one of the selected signs\n    df_selected = df[df['sign'].isin(selected_signs)]\n    # Set random seed for reproducibility\n    np.random.seed(seed)\n    # Sample (participant_size) participants\n    participant_size = train_size + test_size\n    selected_participants = np.random.choice(df_selected['participant_id'], size=participant_size, replace=False)\n    # Return subset of DataFrame with selected participants\n    train_participants = selected_participants[:train_size]\n    test_participants = selected_participants[train_size:participant_size+1]\n    df_train_selected = df_selected[df_selected['participant_id'].isin(train_participants)]\n    df_test_selected = df_selected[df_selected['participant_id'].isin(test_participants)]\n    print(df_train_selected)\n    print(df_test_selected)\n    for sign in selected_signs:\n        one_sign_train = df_train_selected[df_train_selected[\"sign\"]==sign].sample(n=train_size, random_state=42)\n        df_train = pd.concat([df_train, one_sign_train], ignore_index=True)\n        one_sign_test = df_test_selected[df_test_selected[\"sign\"]==sign].sample(n=test_size, random_state=42)\n        df_test = pd.concat([df_test, one_sign_test], ignore_index=True)\n    return df_train , df_test","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-08T11:59:41.244269Z","iopub.execute_input":"2023-03-08T11:59:41.244675Z","iopub.status.idle":"2023-03-08T11:59:41.256001Z","shell.execute_reply.started":"2023-03-08T11:59:41.244641Z","shell.execute_reply":"2023-03-08T11:59:41.254925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nImporting train.csv and creating df_train and df_test with specified signs\n'''\npath = r'/kaggle/input/asl-signs/train.csv'\ndf = pd.read_csv(path)\nsigns = [\"drink\",\"water\",\"after\",\"another\",\"child\",\"dad\",\"every\",\"thankyou\",\"bye\",\"airplane\"]\ndf_train, df_test = select_participants(df,signs,seed=42,train_size=10,test_size=5)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T11:59:44.811399Z","iopub.execute_input":"2023-03-08T11:59:44.811823Z","iopub.status.idle":"2023-03-08T11:59:44.997171Z","shell.execute_reply.started":"2023-03-08T11:59:44.811788Z","shell.execute_reply":"2023-03-08T11:59:44.995772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2023-03-08T11:59:48.276614Z","iopub.execute_input":"2023-03-08T11:59:48.277027Z","iopub.status.idle":"2023-03-08T11:59:48.297942Z","shell.execute_reply.started":"2023-03-08T11:59:48.276992Z","shell.execute_reply":"2023-03-08T11:59:48.296819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''Function loading x,y,z from one parquet file and format into np.shape(n_frames, 75 (number of landmarks), 3 (x,y,z))'''\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z','type']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    data = data[data[\"type\"] != \"face\"]\n    data.drop('type', inplace=True, axis=1)\n    ROWS_PER_FRAME = 75 # number of landmarks per frame\n    data_columns = ['x', 'y', 'z']\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T11:59:50.782062Z","iopub.execute_input":"2023-03-08T11:59:50.782497Z","iopub.status.idle":"2023-03-08T11:59:50.790834Z","shell.execute_reply.started":"2023-03-08T11:59:50.782458Z","shell.execute_reply":"2023-03-08T11:59:50.789309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subset_example = load_relevant_data_subset(r'/kaggle/input/asl-signs/train_landmark_files/16069/100015657.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-03-08T09:04:10.215429Z","iopub.execute_input":"2023-03-08T09:04:10.215896Z","iopub.status.idle":"2023-03-08T09:04:10.41015Z","shell.execute_reply.started":"2023-03-08T09:04:10.215855Z","shell.execute_reply":"2023-03-08T09:04:10.408872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subset_example.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-08T09:04:24.982606Z","iopub.execute_input":"2023-03-08T09:04:24.983412Z","iopub.status.idle":"2023-03-08T09:04:24.991662Z","shell.execute_reply.started":"2023-03-08T09:04:24.983362Z","shell.execute_reply":"2023-03-08T09:04:24.990346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Take the target as well!!","metadata":{}},{"cell_type":"markdown","source":"## Tensor-Approach","metadata":{}},{"cell_type":"markdown","source":"'''Loading all x,y,z of each train_parquet into one Tensor'''\ntrain_df[\"label\"] = train_df[\"sign\"]\nimport tensorflow as tf\nfinal_tensor = tf.zeros((0,0,75,3))\ntrain_paths = pd.DataFrame(df_train[\"path\"], columns=[\"path\"])\nfor index, row in train_paths[0:16].iterrows():\n    root = train_paths['path'][index]\n    pq_path = f'/kaggle/input/asl-signs/{root}'\n    array = load_relevant_data_subset(pq_path)\n    print(array.shape)\n    n_frames = array.shape[0]\n    array = np.reshape(array,(1,n_frames,75,3))\n    maxLen = 100\n    pq_tensor = tf.convert_to_tensor(array)\n    final_tensor = tf.concat([final_tensor, pq_tensor], axis=0)\n\nfinal_tensor","metadata":{"execution":{"iopub.status.busy":"2023-03-08T10:24:13.896975Z","iopub.execute_input":"2023-03-08T10:24:13.897768Z","iopub.status.idle":"2023-03-08T10:24:13.966003Z","shell.execute_reply.started":"2023-03-08T10:24:13.897708Z","shell.execute_reply":"2023-03-08T10:24:13.964198Z"},"jupyter":{"outputs_hidden":true}}},{"cell_type":"code","source":"'''Loading all x,y,z of each train_parquet into one np.array'''\nsigns = [\"drink\",\"water\",\"after\",\"another\",\"child\",\"dad\",\"every\",\"thankyou\",\"bye\",\"airplane\"]\nimport os\nimport json\n\ntrain_paths = pd.DataFrame(df_train[\"path\"], columns=[\"path\"])\nmaxLen = 100\n# set the desired shape\nnew_shape = (maxLen, 75, 3)\nxs = []\nfor i, row in train_paths.iterrows():\n    root = train_paths['path'][index]\n    pq_path = f'/kaggle/input/asl-signs/{root}'\n    data = load_relevant_data_subset(pq_path)\n    n = data.shape[0]\n    if n < maxLen:\n        # compute the amount of padding needed\n        pad_width = [(0, max(0, new_shape[i] - data.shape[i])) for i in range(len(new_shape))]\n\n        # pad the array\n        data = np.pad(data, pad_width, mode='constant')\n\n    n = data.shape[0]\n    if n <= maxLen:\n        xs.append(data)\n\nX = np.array(xs)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T12:05:23.958326Z","iopub.execute_input":"2023-03-08T12:05:23.958828Z","iopub.status.idle":"2023-03-08T12:05:25.414983Z","shell.execute_reply.started":"2023-03-08T12:05:23.958783Z","shell.execute_reply":"2023-03-08T12:05:25.413653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-08T12:06:24.908413Z","iopub.execute_input":"2023-03-08T12:06:24.908816Z","iopub.status.idle":"2023-03-08T12:06:24.916524Z","shell.execute_reply.started":"2023-03-08T12:06:24.90878Z","shell.execute_reply":"2023-03-08T12:06:24.915235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(df_train)","metadata":{"execution":{"iopub.status.busy":"2023-03-08T12:06:43.357812Z","iopub.execute_input":"2023-03-08T12:06:43.358228Z","iopub.status.idle":"2023-03-08T12:06:43.366393Z","shell.execute_reply.started":"2023-03-08T12:06:43.358194Z","shell.execute_reply":"2023-03-08T12:06:43.364807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}