{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-11-09T05:19:30.058982Z","iopub.execute_input":"2023-11-09T05:19:30.059435Z","iopub.status.idle":"2023-11-09T05:19:52.128392Z","shell.execute_reply.started":"2023-11-09T05:19:30.059396Z","shell.execute_reply":"2023-11-09T05:19:52.126578Z"},"jupyter":{"outputs_hidden":true},"collapsed":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pipeline\n- Data processing\n- Data normalization (?)\n- Model training\n- Model inference/prediction\n- Evaluation","metadata":{}},{"cell_type":"markdown","source":"# Data processing\n- read parquet files and target text labels\n- load the dataset (x and y information)\n\nx -> f(x) -> y\n------\n\npose info -> f(x) -> text category","metadata":{}},{"cell_type":"code","source":"# libraries\n\n# loads the dataset and visualizes\nimport json\nimport matplotlib.pyplot as plt\nfrom matplotlib.animation import FuncAnimation\nfrom IPython.display import HTML","metadata":{"execution":{"iopub.status.busy":"2023-11-09T05:32:40.283402Z","iopub.execute_input":"2023-11-09T05:32:40.283815Z","iopub.status.idle":"2023-11-09T05:32:40.289856Z","shell.execute_reply.started":"2023-11-09T05:32:40.283783Z","shell.execute_reply":"2023-11-09T05:32:40.288502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reads the data from the json file\n\ndef flatten_l_o_l(nested_list):\n    \"\"\"Flatten a list of lists into a single list.\n\n    Args:\n        nested_list (list): \n            – A list of lists (or iterables) to be flattened.\n\n    Returns:\n        list: A flattened list containing all items from the input list of lists.\n    \"\"\"\n    return [item for sublist in nested_list for item in sublist]\n\n\ndef print_ln(symbol=\"-\", line_len=110, newline_before=False, newline_after=False):\n    \"\"\"Print a horizontal line of a specified length and symbol.\n\n    Args:\n        symbol (str, optional): \n            – The symbol to use for the horizontal line\n        line_len (int, optional): \n            – The length of the horizontal line in characters\n        newline_before (bool, optional): \n            – Whether to print a newline character before the line\n        newline_after (bool, optional): \n            – Whether to print a newline character after the line\n    \"\"\"\n    if newline_before: print();\n    print(symbol * line_len)\n    if newline_after: print();\n        \n        \ndef read_json_file(file_path):\n    \"\"\"Read a JSON file and parse it into a Python object.\n\n    Args:\n        file_path (str): The path to the JSON file to read.\n\n    Returns:\n        dict: A dictionary object representing the JSON data.\n        \n    Raises:\n        FileNotFoundError: If the specified file path does not exist.\n        ValueError: If the specified file path does not contain valid JSON data.\n    \"\"\"\n    try:\n        # Open the file and load the JSON data into a Python object\n        with open(file_path, 'r') as file:\n            json_data = json.load(file)\n        return json_data\n    except FileNotFoundError:\n        # Raise an error if the file path does not exist\n        raise FileNotFoundError(f\"File not found: {file_path}\")\n    except ValueError:\n        # Raise an error if the file does not contain valid JSON data\n        raise ValueError(f\"Invalid JSON data in file: {file_path}\")\n        \ndef get_sign_df(pq_path, invert_y=True):\n    sign_df = pd.read_parquet(pq_path)\n    \n    # y value is inverted (Thanks @danielpeshkov)\n    if invert_y: sign_df[\"y\"] *= -1 \n        \n    return sign_df\n\n\nROWS_PER_FRAME = 543  # number of landmarks per frame\ndef load_relevant_data_subset(pq_path):\n    data_columns = ['x', 'y', 'z']\n    data = pd.read_parquet(pq_path, columns=data_columns)\n    n_frames = int(len(data) / ROWS_PER_FRAME)\n    data = data.values.reshape(n_frames, ROWS_PER_FRAME, len(data_columns))\n    return data.astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T05:34:05.078856Z","iopub.execute_input":"2023-11-09T05:34:05.081528Z","iopub.status.idle":"2023-11-09T05:34:05.102622Z","shell.execute_reply.started":"2023-11-09T05:34:05.081473Z","shell.execute_reply":"2023-11-09T05:34:05.101041Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualizes data and transforms the coordinates into a video\n\ndef get_hand_points(hand):\n    \"\"\"Return x, y lists of normalized spatial coordinates for each finger in the hand dataframe.\"\"\"\n    def __get_hand_ax(_axis):\n        return [np.nan_to_num(_x) for _x in \n            [hand.iloc[i][_axis] for i in range(5)]+\\\n            [[hand.iloc[i][_axis] for i in range(j, j+4)] for j in range(5, 21, 4)]+\\\n            [hand.iloc[i][_axis] for i in special_pts]]\n    special_pts = [0, 5, 9, 13, 17, 0]\n    return [__get_hand_ax(_ax) for _ax in ['x','y','z']]\n\ndef get_pose_points(pose):\n    \"\"\"\n    Extracts x and y coordinates from the provided dataframe for pose landmarks.\n\n    Args:\n        pose (pandas.DataFrame): Dataframe containing pose landmarks with columns ['x', 'y', 'z', 'visibility', 'presence'].\n\n    Returns:\n        tuple: Two lists of x and y coordinates, respectively.\n\n    \"\"\"\n    def __get_pose_ax(_axis):\n        return [np.nan_to_num(_x) for _x in [\n            [pose.iloc[i][_axis] for i in [8, 6, 5, 4, 0, 1, 2, 3, 7]], \n            [pose.iloc[i][_axis] for i in [10, 9]], \n            [pose.iloc[i][_axis] for i in [22, 16, 20, 18, 16, 14, 12, 11, 13, 15, 17, 19, 15, 21]], \n            [pose.iloc[i][_axis] for i in [12, 24, 26, 28, 30, 32, 28]], \n            [pose.iloc[i][_axis] for i in [11, 23, 25, 27, 29, 31, 27]], \n            [pose.iloc[i][_axis] for i in [24, 23]]\n        ]]\n    return [__get_pose_ax(_ax) for _ax in ['x','y','z']]\n\n\ndef animation_frame(f, event_df, ax, ax_pad=0.2, style=\"full\", \n                    face_color=\"spring\", pose_color=\"autumn\", lh_color=\"winter\", rh_color=\"summer\"):\n    \"\"\"\n    Function called by FuncAnimation to animate the plot with the provided frame.\n\n    Args:\n        f (int): The current frame number.\n\n    Returns:\n        None.\n    \"\"\"\n    \n    face_color = plt.cm.get_cmap(face_color)\n    pose_color = plt.cm.get_cmap(pose_color)\n    rh_color = plt.cm.get_cmap(rh_color)\n    lh_color = plt.cm.get_cmap(lh_color)\n    \n    sign_df = event_df.copy()\n    \n    # Clear axis and fix the axis\n    ax.clear()\n    if style==\"full\":\n        xmin = sign_df['x'].min() - ax_pad\n        xmax = sign_df['x'].max() + ax_pad\n        ymin = sign_df['y'].min() - ax_pad\n        ymax = sign_df['y'].max() + ax_pad\n    elif style==\"hands\":\n        xmin = sign_df[sign_df.type.isin([\"left_hand\", \"right_hand\"])]['x'].min() - ax_pad\n        xmax = sign_df[sign_df.type.isin([\"left_hand\", \"right_hand\"])]['x'].max() + ax_pad\n        ymin = sign_df[sign_df.type.isin([\"left_hand\", \"right_hand\"])]['y'].min() - ax_pad\n        ymax = sign_df[sign_df.type.isin([\"left_hand\", \"right_hand\"])]['y'].max() + ax_pad\n    else:\n        xmin = sign_df[sign_df.type==style]['x'].min() - ax_pad\n        xmax = sign_df[sign_df.type==style]['x'].max() + ax_pad\n        ymin = sign_df[sign_df.type==style]['y'].min() - ax_pad\n        ymax = sign_df[sign_df.type==style]['y'].max() + ax_pad\n    \n    ax.set_xlim(xmin, xmax)\n    ax.set_ylim(ymin, ymax)\n    ax.axis(False) # Remove the axis lines\n    \n    # Normalize depth\n    zmin, zmax = sign_df['z'].min(), sign_df['z'].max()\n    sign_df['z'] = (sign_df['z']-zmin)/(zmax-zmin)\n    \n    # Get data for current frame\n    frame = sign_df[sign_df.frame==f]\n    \n    # Left Hand\n    if style.lower() in [\"left_hand\", \"hands\", \"full\"]:\n        left = frame[frame.type=='left_hand']\n        lx, ly, lz = get_hand_points(left)\n        for i in range(len(lx)):\n            if type(lx[i])!=np.float64:\n                lh_clr = [lh_color(((np.abs(_x)+np.abs(_y))/2)) for _x, _y in zip(lx[i], ly[i])]\n                lh_clr = tuple(sum(_x)/len(_x) for _x in zip(*lh_clr))\n            else: \n                lh_clr = lh_color(((np.abs(lx[i])+np.abs(ly[i]))/2))\n            ax.plot(lx[i], ly[i], color=lh_clr, alpha=lz[i].mean())\n    \n    # Right Hand\n    if style.lower() in [\"right_hand\", \"hands\", \"full\"]:\n        right = frame[frame.type=='right_hand']\n        rx, ry, rz = get_hand_points(right)\n        for i in range(len(rx)):\n            if type(rx[i])!=np.float64:\n                rh_clr = [rh_color((np.abs(_x)+np.abs(_y))/2) for _x, _y in zip(rx[i], ry[i])] \n                rh_clr = tuple(sum(_x)/len(_x) for _x in zip(*rh_clr))\n            else:\n                rh_clr = rh_color(((np.abs(rx[i])+np.abs(ry[i]))/2))\n            ax.plot(rx[i], ry[i], color=rh_clr, alpha=rz[i].mean())\n    \n    # Pose\n    if style.lower() in [\"pose\", \"full\"]:\n        pose = frame[frame.type=='pose']\n        px, py, pz = get_pose_points(pose)\n        for i in range(len(px)):\n            if type(px[i])!=np.float64:\n                pose_clr = [pose_color(((np.abs(_x)+np.abs(_y))/2)) for _x, _y in zip(px[i], py[i])]\n                pose_clr = tuple(sum(_x)/len(_x) for _x in zip(*pose_clr))\n            else: \n                pose_clr = pose_color(((np.abs(px[i])+np.abs(py[i]))/2))\n            ax.plot(px[i], py[i], color=pose_clr, alpha=pz[i].mean())\n        \n    if style.lower() in [\"face\", \"full\"]:\n        face = frame[frame.type=='face'][['x', 'y', 'z']].values\n        fx, fy, fz = face[:,0], face[:,1], face[:,2]\n        for i in range(len(fx)):\n            ax.plot(fx[i], fy[i], '.', color=pose_color(fz[i]), alpha=fz[i])\n    \n    # Use this so we don't get an extra return\n    plt.close()\n    \n    \ndef plot_event(event_df, style=\"full\"):\n    # Create figure and animation\n    fig, ax = plt.subplots()\n    l, = ax.plot([], [])\n    animation = FuncAnimation(fig, func=lambda x: animation_frame(x, event_df, ax, style=style), \n                              frames=event_df[\"frame\"].unique())\n    \n    # Display animation as HTML5 video\n    return HTML(animation.to_html5_video())","metadata":{"execution":{"iopub.status.busy":"2023-11-09T05:34:33.164978Z","iopub.execute_input":"2023-11-09T05:34:33.165401Z","iopub.status.idle":"2023-11-09T05:34:33.208989Z","shell.execute_reply.started":"2023-11-09T05:34:33.165369Z","shell.execute_reply":"2023-11-09T05:34:33.207564Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Loading the parquet files\n\n# Define the path to the root data directory\nDATA_DIR         = \"/kaggle/input/asl-signs\"\nEXTEND_TRAIN_DIR = \"/kaggle/input/gislr-extended-train-dataframe\" \n\nprint(\"\\n... BASIC DATA SETUP STARTING ...\\n\")\nprint(\"\\n\\n... LOAD TRAIN DATAFRAME FROM CSV FILE ...\\n\")\n\n\ntrain_df = pd.read_csv(os.path.join(DATA_DIR, \"train.csv\"))\ntrain_df[\"path\"] = DATA_DIR+\"/\"+train_df[\"path\"]\ndisplay(train_df)\n\nprint(\"\\n\\n... LOAD SIGN TO PREDICTION INDEX MAP FROM JSON FILE ...\\n\")\ns2p_map = {k.lower():v for k,v in read_json_file(os.path.join(DATA_DIR, \"sign_to_prediction_index_map.json\")).items()}\np2s_map = {v:k for k,v in read_json_file(os.path.join(DATA_DIR, \"sign_to_prediction_index_map.json\")).items()}\nencoder = lambda x: s2p_map.get(x.lower())\ndecoder = lambda x: p2s_map.get(x)\nprint(s2p_map)\n\nDEMO_ROW = 0\nprint(f\"\\n\\n... DEMO SIGN/EVENT DATAFRAME FOR ROW {DEMO_ROW} - SIGN={train_df.iloc[DEMO_ROW]['sign']} ...\\n\")\ndemo_sign_df = get_sign_df(train_df.iloc[DEMO_ROW][\"path\"])\ndisplay(demo_sign_df)\n\n# I messed this function up... will fix later\nplot_event(demo_sign_df)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T05:56:02.719276Z","iopub.execute_input":"2023-11-09T05:56:02.719670Z","iopub.status.idle":"2023-11-09T05:56:22.932732Z","shell.execute_reply.started":"2023-11-09T05:56:02.719642Z","shell.execute_reply":"2023-11-09T05:56:22.931391Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data loader\n\nimport torch\n\ndef load_one_batch(df):\n    frames = df['frame'].values\n    x_values = df['x'].values\n    y_values = df['y'].values\n    z_values = df['z'].values\n\n    # Stack the x, y, and z values together into a single tensor\n    data = torch.tensor([frames, x_values, y_values, z_values], dtype=torch.float)\n\n    # Reshape the tensor into the desired shape [1, frame, x, y, z]\n    data = data.view(1, -1, 1, 1, 1)\n    return data\n\nx = load_one_batch(demo_sign_df)\ny = torch.Tensor(0)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T06:25:37.294055Z","iopub.execute_input":"2023-11-09T06:25:37.295068Z","iopub.status.idle":"2023-11-09T06:25:37.325717Z","shell.execute_reply.started":"2023-11-09T06:25:37.295029Z","shell.execute_reply":"2023-11-09T06:25:37.324456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-09T06:25:38.564208Z","iopub.execute_input":"2023-11-09T06:25:38.564667Z","iopub.status.idle":"2023-11-09T06:25:38.572754Z","shell.execute_reply.started":"2023-11-09T06:25:38.564632Z","shell.execute_reply":"2023-11-09T06:25:38.571536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load data x and y","metadata":{"execution":{"iopub.status.busy":"2023-11-09T06:25:27.824208Z","iopub.execute_input":"2023-11-09T06:25:27.824620Z","iopub.status.idle":"2023-11-09T06:25:27.829199Z","shell.execute_reply.started":"2023-11-09T06:25:27.824592Z","shell.execute_reply":"2023-11-09T06:25:27.828162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dummy data for model training","metadata":{}},{"cell_type":"code","source":"# shape: [batchsize, frames, x, y, z]\nx = torch.randn(4, 80364, 1, 1, 1)\nx.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-09T06:26:51.314221Z","iopub.execute_input":"2023-11-09T06:26:51.315527Z","iopub.status.idle":"2023-11-09T06:26:51.327352Z","shell.execute_reply.started":"2023-11-09T06:26:51.315485Z","shell.execute_reply":"2023-11-09T06:26:51.325808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = torch.Tensor([0, 11, 9, 0])\ny.shape","metadata":{"execution":{"iopub.status.busy":"2023-11-09T06:26:52.315572Z","iopub.execute_input":"2023-11-09T06:26:52.316356Z","iopub.status.idle":"2023-11-09T06:26:52.325044Z","shell.execute_reply.started":"2023-11-09T06:26:52.316312Z","shell.execute_reply":"2023-11-09T06:26:52.324061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the dataset for pytorch\n# subsets the dataset per batch size\n\n# Define batch size\nbatch_size = 2  # Adjust the batch size according to your needs\n\nimport torch\nfrom torch.utils.data import TensorDataset, DataLoader\n\n# Create a TensorDataset from x and y\ndataset = TensorDataset(x, y)\n\n# Create a DataLoader\ndataloader = DataLoader(dataset, batch_size=batch_size, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-11-09T06:26:53.521114Z","iopub.execute_input":"2023-11-09T06:26:53.522124Z","iopub.status.idle":"2023-11-09T06:26:53.528732Z","shell.execute_reply.started":"2023-11-09T06:26:53.522083Z","shell.execute_reply":"2023-11-09T06:26:53.527409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# You can iterate through the dataloader in your training loop\n# visualizing the subsets for a sanity check\nfor batch_x, batch_y in dataloader:\n    print(f\"x tensor (videos): {batch_x.shape}\")\n    print(f\"y tensor (text category) {batch_y.shape}\")\n    break\n    # Your training code here","metadata":{"execution":{"iopub.status.busy":"2023-11-09T06:27:00.661996Z","iopub.execute_input":"2023-11-09T06:27:00.662425Z","iopub.status.idle":"2023-11-09T06:27:00.672352Z","shell.execute_reply.started":"2023-11-09T06:27:00.662394Z","shell.execute_reply":"2023-11-09T06:27:00.670976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model training\n- x, y\n- Find f(x) which can predict y from x inputs\n","metadata":{}},{"cell_type":"code","source":"# basic model from chatgpt\n\nimport torch\nimport torch.nn as nn\n\n# Define your neural network model\nclass BasicModel(nn.Module):\n    def __init__(self, input_size, hidden_size, num_classes):\n        super(BasicModel, self).__init__()\n        self.fc1 = nn.Linear(input_size, hidden_size)\n        self.relu = nn.ReLU()\n        self.fc2 = nn.Linear(hidden_size, num_classes)\n\n    def forward(self, x):\n        x = x.view(x.size(0), -1)  # Flatten the input tensor if necessary\n        x = self.fc1(x)\n        x = self.relu(x)\n        x = self.fc2(x)\n        return x\n\n# Define the input and output sizes\ninput_size = 80364  # frames\nhidden_size = 128  # model size\nnum_classes = 250  # number of categories (text labels from the json file)\n\n# Create an instance of the model\nmodel = BasicModel(input_size, hidden_size, num_classes)\n\n# You can print the model to see its architecture\nprint(model)\n\n# Define a loss function and optimizer for training\n\n# loss function\ncriterion = nn.CrossEntropyLoss()\n\n# backpropagation\noptimizer = torch.optim.SGD(model.parameters(), lr=0.01) # lr -> learning rate","metadata":{"execution":{"iopub.status.busy":"2023-11-09T06:30:23.053912Z","iopub.execute_input":"2023-11-09T06:30:23.054369Z","iopub.status.idle":"2023-11-09T06:30:23.188940Z","shell.execute_reply.started":"2023-11-09T06:30:23.054335Z","shell.execute_reply":"2023-11-09T06:30:23.187659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training loop\ndef train_model(model, dataloader, criterion, optimizer, num_epochs):\n    for epoch in range(num_epochs):\n        model.train()  # Set the model to training mode\n        total_loss = 0.0\n        correct_predictions = 0\n        total_samples = 0\n        \n        # load each batch from the dataset\n        for inputs, labels in dataloader:\n            # inputs -> videos\n            # labels/outputs -> text category\n            \n            # Zero the gradients\n            optimizer.zero_grad()\n\n            # Forward pass (initial prediction)\n            outputs = model(inputs)\n\n            # Calculate the loss from the loss function\n            loss = criterion(outputs, labels.long())\n\n            # Backpropagation and optimization\n            loss.backward()\n            optimizer.step()\n\n            # Track loss and accuracy\n            total_loss += loss.item()\n            _, predicted = torch.max(outputs, 1)\n            correct_predictions += (predicted == labels).sum().item()\n            total_samples += labels.size(0)\n\n        # Calculate and print statistics for this epoch\n        epoch_loss = total_loss / len(dataloader)\n        accuracy = (correct_predictions / total_samples) * 100.0\n        print(f'Epoch [{epoch + 1}/{num_epochs}] - Loss: {epoch_loss:.4f}, Accuracy: {accuracy:.2f}%')\n\n    print('Training completed.')\n\n# Example usage\nnum_epochs = 10  # Adjust the number of training epochs as needed\ntrain_model(model, dataloader, criterion, optimizer, num_epochs)\n","metadata":{"execution":{"iopub.status.busy":"2023-11-09T06:31:35.481955Z","iopub.execute_input":"2023-11-09T06:31:35.482453Z","iopub.status.idle":"2023-11-09T06:31:36.159766Z","shell.execute_reply.started":"2023-11-09T06:31:35.482412Z","shell.execute_reply":"2023-11-09T06:31:36.158402Z"},"trusted":true},"execution_count":null,"outputs":[]}]}