{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n'''\nFunction selecting train and test parquets from the train.csv\n'''\ndef select_participants(df, selected_signs, seed, participant_size):\n    df_final = pd.DataFrame()\n    # Select rows with one of the selected signs\n    df_selected = df[df['sign'].isin(selected_signs)]\n    # Set random seed for reproducibility\n    np.random.seed(seed)\n    # Sample (participant_size) participants\n    selected_participants = np.random.choice(df_selected['participant_id'], size=participant_size, replace=False)\n    # Return subset of DataFrame with selected participants\n    df_selected = df_selected[df_selected['participant_id'].isin(selected_participants)]\n    for sign in selected_signs:\n        one_sign_df = df_selected[df_selected[\"sign\"]==sign].sample(n=participant_size, random_state=42)\n        df_final = pd.concat([df_final, one_sign_df], ignore_index=True)\n    return df_final\n\n\n'''\nImporting train.csv and creating df_train and df_test with specified signs\n'''\n\npath = r'/kaggle/input/asl-signs/train.csv'\ndf_read = pd.read_csv(path)\nsigns = [\"drink\",\"water\",\"after\",\"another\",\"child\",\"dad\",\"every\",\"thankyou\",\"bye\",\"airplane\"]\ndf = select_participants(df_read,signs,seed=42,participant_size=15)\n\n\n'''\nFunction loading x,y,z from one parquet file and format into np.shape(n_frames, 75 (number of landmarks), 3 (x,y,z))\n'''\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z','type']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    data = data[data[\"type\"] != \"face\"]\n    data.drop('type', inplace=True, axis=1)\n    ROWS_PER_FRAME = 75 # number of landmarks per frame\n    data_columns = ['x', 'y', 'z']\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)\n\n\n'''Loading all x,y,z of each train_parquet into one np.array'''\n\npaths = pd.DataFrame(df[\"path\"], columns=[\"path\"])\nmaxLen = 299\n# set the desired shape\nnew_shape = (maxLen, 75, 3)\nxs = []\nys = []\nn_list = []\nfor i, row in paths.iterrows():\n    root = paths['path'][i]\n    pq_path = f'/kaggle/input/asl-signs/{root}'\n    data = load_relevant_data_subset(pq_path)\n    n = data.shape[0]\n    if n < maxLen:\n        # compute the amount of padding needed\n        pad_width = [(0, max(0, new_shape[i] - data.shape[i])) for i in range(len(new_shape))]\n\n        # pad the array\n        data = np.pad(data, pad_width, mode='constant',constant_values=(999,))\n        \n    n = data.shape[0]\n    n_list.append(n)\n    if n <= maxLen:\n        xs.append(data)\n        ys.append(df[\"sign\"][i])\n\nX = np.array(xs)\ny = np.array(ys)\n\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-08T14:23:39.991933Z","iopub.execute_input":"2023-03-08T14:23:39.992341Z","iopub.status.idle":"2023-03-08T14:23:41.832818Z","shell.execute_reply.started":"2023-03-08T14:23:39.992306Z","shell.execute_reply":"2023-03-08T14:23:41.831931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-08T14:23:43.711224Z","iopub.execute_input":"2023-03-08T14:23:43.712820Z","iopub.status.idle":"2023-03-08T14:23:43.726333Z","shell.execute_reply.started":"2023-03-08T14:23:43.712734Z","shell.execute_reply":"2023-03-08T14:23:43.723300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def select_participants(df, selected_signs, seed, participant_size):\n    df_final = pd.DataFrame()\n    # Select rows with one of the selected signs\n    df_selected = df[df['sign'].isin(selected_signs)]\n    # Set random seed for reproducibility\n    np.random.seed(seed)\n    # Sample (participant_size) participants\n    selected_participants = np.random.choice(df_selected['participant_id'], size=participant_size, replace=False)\n    # Return subset of DataFrame with selected participants\n    df_selected = df_selected[df_selected['participant_id'].isin(selected_participants)]\n    for sign in selected_signs:\n        one_sign_df = df_selected[df_selected[\"sign\"]==sign].sample(n=participant_size, random_state=42)\n        df_final = pd.concat([df_final, one_sign_df], ignore_index=True)\n    return df_final\n\n\n'''\nImporting train.csv and creating df_train and df_test with specified signs\n'''\n\npath = r'/kaggle/input/asl-signs/train.csv'\ndf_read = pd.read_csv(path)\nsigns = [\"drink\",\"water\",\"after\",\"another\",\"child\",\"dad\",\"every\",\"thankyou\",\"bye\",\"airplane\"]\ndf = select_participants(df_read,signs,seed=42,participant_size=15)","metadata":{},"execution_count":null,"outputs":[]}]}