{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":130287,"databundleVersionId":15633993},{"sourceType":"datasetVersion","sourceId":15314803,"datasetId":9791156,"databundleVersionId":16220261,"isSourceIdPinned":true},{"sourceType":"datasetVersion","sourceId":15748815,"datasetId":10092016,"databundleVersionId":16691857},{"sourceType":"datasetVersion","sourceId":15268252,"datasetId":9765169,"databundleVersionId":16167997},{"sourceType":"modelInstanceVersion","sourceId":740844,"databundleVersionId":15575334,"modelInstanceId":558490,"modelId":571057}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport numpy as np\nimport pandas as pd\nimport os\nimport tqdm\n\nimport numpy as np\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport argparse\nfrom typing import Tuple, List, Optional","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:55:19.580959Z","iopub.execute_input":"2026-04-16T13:55:19.581445Z","iopub.status.idle":"2026-04-16T13:55:23.896426Z","shell.execute_reply.started":"2026-04-16T13:55:19.581405Z","shell.execute_reply":"2026-04-16T13:55:23.895465Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!kaggle kernels output antonygithinji/motion-s-dataset-exploration -p /kaggle/working/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:55:23.898831Z","iopub.execute_input":"2026-04-16T13:55:23.899423Z","iopub.status.idle":"2026-04-16T13:55:25.831633Z","shell.execute_reply.started":"2026-04-16T13:55:23.899391Z","shell.execute_reply":"2026-04-16T13:55:25.830145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"rvq_vae_path = \"/kaggle/input/models/antonygithinji/motion-s-vae-rvq/pytorch/default/3/rvq_vae_best.pth\"\n\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:55:25.833697Z","iopub.execute_input":"2026-04-16T13:55:25.835189Z","iopub.status.idle":"2026-04-16T13:55:25.842023Z","shell.execute_reply.started":"2026-04-16T13:55:25.835139Z","shell.execute_reply":"2026-04-16T13:55:25.840878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission = pd.read_csv(\"/kaggle/input/datasets/maksimrazantsau/my-motion-checkpoints/submission.csv\")\nsubmission_head = submission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:55:25.843401Z","iopub.execute_input":"2026-04-16T13:55:25.843808Z","iopub.status.idle":"2026-04-16T13:55:26.009006Z","shell.execute_reply.started":"2026-04-16T13:55:25.843769Z","shell.execute_reply":"2026-04-16T13:55:26.008123Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission_head","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:55:26.010741Z","iopub.execute_input":"2026-04-16T13:55:26.011169Z","iopub.status.idle":"2026-04-16T13:55:26.057679Z","shell.execute_reply.started":"2026-04-16T13:55:26.011131Z","shell.execute_reply":"2026-04-16T13:55:26.056804Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from torchsummary import summary  #for model summary\nclass MotionEncoder(nn.Module):\n    \"\"\"\n    1D Convolutional Encoder for motion sequences.\n    Input: (B, D_POSE, N) - treats D_POSE as channels, N as sequence length.\n    Output: (B, d, n) - downsampled latent with n = N // downsampling_ratio.\n    Uses stacked Conv1D with stride=downsampling_factor for downsampling.\n    \"\"\"\n    def __init__(self, input_dim, latent_dim=256, downsampling_ratio=4, num_layers=4, hidden_dim=512):\n        super(MotionEncoder, self).__init__()\n        self.input_dim = input_dim\n        self.latent_dim = latent_dim\n        self.downsampling_ratio = downsampling_ratio\n        self.stride = 2  # Assuming binary downsampling (stride=2 per layer)\n        self.num_layers = num_layers\n        \n        # Build conv layers\n        layers = []\n        current_dim = input_dim\n        for i in range(num_layers):\n            out_dim = hidden_dim if i < num_layers - 1 else latent_dim\n            stride = self.stride if i < int(np.log2(downsampling_ratio)) else 1\n            layers.extend([\n                nn.Conv1d(current_dim, out_dim, kernel_size=3, stride=stride, padding=1),\n                nn.ReLU(inplace=True),\n                nn.BatchNorm1d(out_dim)  # Stabilizes training\n            ])\n            current_dim = out_dim\n        \n        self.encoder = nn.Sequential(*layers)\n        self.downsampled_len = None  # Computed on first forward\n    \n    def forward(self, x):\n        # x: (B, D_POSE, N)\n        if self.downsampled_len is None:\n            with torch.no_grad():\n                dummy = self.encoder(x)\n                self.downsampled_len = dummy.shape[-1]\n        \n        encoded = self.encoder(x)  # (B, latent_dim, n)\n        return encoded  # Keep as (B, d, n); can transpose/reshape downstream if needed\n\n\nclass MotionDecoder(nn.Module):\n    \"\"\"\n    1D Convolutional Decoder for motion sequences.\n    Input: (B, d, n) - downsampled latent with n = N // downsampling_ratio.\n    Output: (B, D_POSE, N) - upsampled to original motion dimension.\n    Uses stacked Conv1D with transpose convolutions for upsampling.\n    \"\"\"\n    def __init__(self, output_dim, latent_dim=256, downsampling_ratio=4, num_layers=4, hidden_dim=512):\n        super(MotionDecoder, self).__init__()\n        self.output_dim = output_dim\n        self.latent_dim = latent_dim\n        self.downsampling_ratio = downsampling_ratio\n        self.stride = 2  # Binary upsampling (stride=2 per layer)\n        self.num_layers = num_layers\n        \n        # Build conv transpose layers (mirror of encoder)\n        layers = []\n        current_dim = latent_dim\n        for i in range(num_layers):\n            out_dim = hidden_dim if i < num_layers - 1 else output_dim\n            stride = self.stride if i < int(np.log2(downsampling_ratio)) else 1\n            # Use ConvTranspose1d for upsampling\n            # output_padding compensates for stride to ensure exact upsampling\n            output_padding = (stride - 1) if stride > 1 else 0\n            layers.extend([\n                nn.ConvTranspose1d(current_dim, out_dim, kernel_size=3, stride=stride, padding=1, output_padding=output_padding),\n                nn.ReLU(inplace=True) if i < num_layers - 1 else nn.Identity(),\n                nn.BatchNorm1d(out_dim) if i < num_layers - 1 else nn.Identity()\n            ])\n            current_dim = out_dim\n        \n        self.decoder = nn.Sequential(*layers)\n    \n    def forward(self, x):\n        # x: (B, latent_dim, n)\n        decoded = self.decoder(x)  # (B, D_POSE, N)\n        return decoded\n\n\nclass VectorQuantizer(nn.Module):\n    \"\"\"\n    Single-layer vector quantizer with EMA updates.\n    Implements straight-through estimator for gradients.\n    \"\"\"\n    def __init__(self, num_embeddings=512, embedding_dim=256, commitment_cost=1.0, \n                 decay=0.99, epsilon=1e-5):\n        super(VectorQuantizer, self).__init__()\n        self.embedding_dim = embedding_dim\n        self.num_embeddings = num_embeddings\n        self.commitment_cost = commitment_cost\n        self.decay = decay\n        self.epsilon = epsilon\n        \n        # Initialize codebook\n        self.register_buffer('embedding', torch.randn(embedding_dim, num_embeddings))\n        self.register_buffer('cluster_size', torch.zeros(num_embeddings))\n        self.register_buffer('embedding_avg', torch.zeros(embedding_dim, num_embeddings))\n        \n    def forward(self, inputs):\n        # inputs: (B, d, n) - reshape to (B*n, d) for quantization\n        B, d, n = inputs.shape\n        flat_input = inputs.permute(0, 2, 1).contiguous()  # (B, n, d)\n        flat_input = flat_input.reshape(-1, d)  # (B*n, d)\n        \n        # Calculate distances to codebook entries\n        distances = (torch.sum(flat_input**2, dim=1, keepdim=True) \n                    + torch.sum(self.embedding**2, dim=0, keepdim=True)\n                    - 2 * torch.matmul(flat_input, self.embedding))  # (B*n, num_embeddings)\n        \n        # Find nearest codebook entry\n        encoding_indices = torch.argmin(distances, dim=1)  # (B*n,)\n        \n        # Quantize: replace with nearest codebook entry\n        quantized = self.embedding[:, encoding_indices].t()  # (B*n, d)\n        quantized = quantized.reshape(B, n, d).permute(0, 2, 1)  # (B, d, n)\n        \n        # Straight-through estimator: use quantized in forward, gradients pass through inputs\n        quantized_st = inputs + (quantized - inputs).detach()\n        \n        # Commitment loss: encourage inputs to be close to quantized codes\n        e_latent_loss = F.mse_loss(quantized.detach(), inputs)\n        loss = self.commitment_cost * e_latent_loss\n        \n        # EMA update (only during training)\n        # EMA update (only during training)\n        if self.training:\n            with torch.no_grad():  # Critical: wrap entire EMA update\n                # Update cluster sizes and embedding averages\n                encodings = F.one_hot(encoding_indices, self.num_embeddings).float()  # (B*n, num_embeddings)\n                \n                # Detach flat_input before using in EMA updates\n                flat_input_detached = flat_input.detach()\n                \n                # Update cluster size\n                self.cluster_size.mul_(self.decay).add_(encodings.sum(0), alpha=1 - self.decay)\n                \n                # Update embedding averages (using detached input)\n                embed_sum = flat_input_detached.t() @ encodings  # (d, num_embeddings)\n                self.embedding_avg.mul_(self.decay).add_(embed_sum, alpha=1 - self.decay)\n                \n                # Update codebook entries\n                n_clusters = self.cluster_size.sum()\n                cluster_size = (\n                    (self.cluster_size + self.epsilon) / \n                    (n_clusters + self.num_embeddings * self.epsilon) * \n                    n_clusters\n                )\n                embed_normalized = self.embedding_avg / cluster_size.unsqueeze(0)\n                self.embedding.copy_(embed_normalized)\n        \n        # Reshape encoding_indices back to (B, n)\n        encoding_indices = encoding_indices.reshape(B, n)\n        \n        return quantized_st, encoding_indices, loss\n\n\nclass ResidualVectorQuantizer(nn.Module):\n    \"\"\"\n    Residual Vector Quantization (RVQ) with V+1 quantization layers.\n    Implements hierarchical quantization as described in MoMask paper.\n    \"\"\"\n    def __init__(self, num_quantizers=6, num_embeddings=512, embedding_dim=256, \n                 commitment_cost=1.0, decay=0.99, quantization_dropout=0.2):\n        super(ResidualVectorQuantizer, self).__init__()\n        self.num_quantizers = num_quantizers  # V+1 total layers (including base)\n        self.quantization_dropout = quantization_dropout\n        \n        # Create V+1 quantizers\n        self.quantizers = nn.ModuleList([\n            VectorQuantizer(\n                num_embeddings=num_embeddings,\n                embedding_dim=embedding_dim,\n                commitment_cost=commitment_cost,\n                decay=decay\n            ) for _ in range(num_quantizers)\n        ])\n    \n    def forward(self, inputs, return_tokens=True):\n        \"\"\"\n        Forward pass through residual quantization.\n        \n        Args:\n            inputs: (B, d, n) - continuous latent sequence from encoder\n            return_tokens: if True, return token indices; if False, only return quantized codes\n        \n        Returns:\n            quantized: (B, d, n) - sum of all quantized codes\n            tokens: List of (B, n) token sequences, one per layer\n            commitment_loss: scalar - sum of commitment losses from all layers\n        \"\"\"\n        B, d, n = inputs.shape\n        \n        # Initialize residual with input\n        residual = inputs  # r^0 = b̃\n        quantized_codes = []\n        tokens = []\n        commitment_losses = []\n        \n        # Determine how many layers to use (quantization dropout during training)\n        num_active_layers = self.num_quantizers\n        if self.training and self.quantization_dropout > 0:\n            # Randomly disable last 0 to V layers\n            if torch.rand(1).item() < self.quantization_dropout:\n                num_active_layers = torch.randint(0, self.num_quantizers, (1,)).item() + 1\n        \n        # Quantize through each layer\n        for v in range(num_active_layers):\n            # Quantize current residual: b^v = Q(r^v)\n            quantized, token_indices, loss = self.quantizers[v](residual)\n            \n            quantized_codes.append(quantized)\n            tokens.append(token_indices)\n            commitment_losses.append(loss)\n            \n            # Compute next residual: r^{v+1} = r^v - b^v\n            if v < num_active_layers - 1:\n                residual = residual - quantized\n        \n        # Pad with zeros if fewer layers were used\n        while len(quantized_codes) < self.num_quantizers:\n            quantized_codes.append(torch.zeros_like(inputs))\n            tokens.append(torch.zeros((B, n), dtype=torch.long, device=inputs.device))\n        \n        # Sum all quantized codes: Σ_{v=0}^V b^v\n        quantized = sum(quantized_codes)\n        \n        # Total commitment loss\n        commitment_loss = sum(commitment_losses)\n        \n        if return_tokens:\n            return quantized, tokens, commitment_loss\n        else:\n            return quantized, commitment_loss\n    \n    def quantize_from_tokens(self, tokens):\n        \"\"\"\n        Reconstruct quantized codes from token indices.\n        \n        Args:\n            tokens: List of (B, n) token sequences, one per layer\n        \n        Returns:\n            quantized: (B, d, n) - sum of all quantized codes\n        \"\"\"\n        quantized_codes = []\n        \n        for v, token_seq in enumerate(tokens):\n            B, n = token_seq.shape\n            d = self.quantizers[v].embedding.shape[0]\n            \n            # Lookup codebook entries\n            # token_seq: (B, n) with indices in [0, num_embeddings-1]\n            # embedding: (d, num_embeddings)\n            # We want: (B, d, n)\n            flat_tokens = token_seq.reshape(-1)  # (B*n,)\n            quantized_flat = self.quantizers[v].embedding[:, flat_tokens]  # (d, B*n)\n            quantized = quantized_flat.reshape(d, B, n).permute(1, 0, 2)  # (B, d, n)\n            \n            quantized_codes.append(quantized)\n        \n        # Sum all quantized codes\n        quantized = sum(quantized_codes)\n        return quantized\n\n\nclass RVQVAE(nn.Module):\n    \"\"\"\n    Complete Residual VQ-VAE model combining encoder, RVQ, and decoder.\n    \"\"\"\n    def __init__(self, input_dim, output_dim, latent_dim=256, downsampling_ratio=4,\n                 num_layers=4, hidden_dim=512, num_quantizers=6, num_embeddings=512,\n                 commitment_cost=1.0, decay=0.99, quantization_dropout=0.2):\n        super(RVQVAE, self).__init__()\n        \n        self.encoder = MotionEncoder(\n            input_dim=input_dim,\n            latent_dim=latent_dim,\n            downsampling_ratio=downsampling_ratio,\n            num_layers=num_layers,\n            hidden_dim=hidden_dim\n        )\n        \n        self.rvq = ResidualVectorQuantizer(\n            num_quantizers=num_quantizers,\n            num_embeddings=num_embeddings,\n            embedding_dim=latent_dim,\n            commitment_cost=commitment_cost,\n            decay=decay,\n            quantization_dropout=quantization_dropout\n        )\n        \n        self.decoder = MotionDecoder(\n            output_dim=output_dim,\n            latent_dim=latent_dim,\n            downsampling_ratio=downsampling_ratio,\n            num_layers=num_layers,\n            hidden_dim=hidden_dim\n        )\n        \n        self.input_dim = input_dim\n        self.output_dim = output_dim\n        self.latent_dim = latent_dim\n        self.downsampling_ratio = downsampling_ratio\n        self.num_quantizers = num_quantizers\n    \n    def forward(self, x, return_tokens=True):\n        \"\"\"\n        Forward pass through RVQ-VAE.\n        \n        Args:\n            x: (B, D_POSE, N) - input motion sequence\n            return_tokens: if True, return token sequences\n        \n        Returns:\n            reconstructed: (B, D_POSE, N) - reconstructed motion\n            tokens: List of (B, n) token sequences (if return_tokens=True)\n            commitment_loss: scalar - commitment loss from RVQ\n        \"\"\"\n        # Encode\n        latent = self.encoder(x)  # (B, d, n)\n        \n        # Quantize\n        quantized, tokens, commitment_loss = self.rvq(latent, return_tokens=return_tokens)\n        \n        # Decode\n        reconstructed = self.decoder(quantized)  # (B, D_POSE, N)\n        \n        if return_tokens:\n            return reconstructed, tokens, commitment_loss\n        else:\n            return reconstructed, commitment_loss\n    \n    def encode_to_tokens(self, x):\n        \"\"\"\n        Encode motion to discrete tokens without decoding.\n        \n        Args:\n            x: (B, D_POSE, N) - input motion sequence\n        \n        Returns:\n            tokens: List of (B, n) token sequences\n        \"\"\"\n        latent = self.encoder(x)  # (B, d, n)\n        _, tokens, _ = self.rvq(latent, return_tokens=True)\n        return tokens\n    \n    def decode_from_tokens(self, tokens):\n        \"\"\"\n        Decode discrete tokens back to motion.\n        \n        Args:\n            tokens: List of (B, n) token sequences\n        \n        Returns:\n            reconstructed: (B, D_POSE, N) - reconstructed motion\n        \"\"\"\n        quantized = self.rvq.quantize_from_tokens(tokens)  # (B, d, n)\n        reconstructed = self.decoder(quantized)  # (B, D_POSE, N)\n        return reconstructed\n\n\ndef compute_output_length(model, seq_len):\n    \"\"\"\n    Compute the actual downsampled length by running a dummy forward on CPU.\n    \"\"\"\n    model.eval()\n    model.to('cpu')\n    dummy_input = torch.randn(1, model.input_dim, seq_len)\n    with torch.no_grad():\n        output = model(dummy_input)\n        return output.shape[-1]\n\ndef encode_motion(features, model, device='cuda' if torch.cuda.is_available() else 'cpu'):\n    \"\"\"\n    Encode the (N, D_POSE) features to latent (n, d).\n    Returns: latent_np (n, d) as numpy array.\n    \"\"\"\n    model.eval()\n    model.to(device)\n    \n    # Prepare input: (1, D_POSE, N)\n    N, D = features.shape\n    x = torch.from_numpy(features.T).unsqueeze(0).float().to(device)  # (1, D, N)\n    \n    with torch.no_grad():\n        if isinstance(model, RVQVAE):\n            # For RVQVAE, encode to tokens first, then get quantized latents\n            tokens = model.encode_to_tokens(x)\n            quantized = model.rvq.quantize_from_tokens(tokens)\n            latent_np = quantized.squeeze(0).cpu().numpy().T  # (n, d)\n        else:\n            # For regular encoder\n            latent = model(x)  # (1, d, n)\n            latent_np = latent.squeeze(0).cpu().numpy().T  # (n, d)\n    \n    return latent_np\n\n\ndef decode_motion(latent, model, device='cuda' if torch.cuda.is_available() else 'cpu'):\n    \"\"\"\n    Decode the (n, d) latent to motion features (N, D_POSE).\n    Returns: motion_np (N, D_POSE) as numpy array.\n    \"\"\"\n    model.eval()\n    model.to(device)\n    \n    # Prepare input: (1, d, n)\n    n, d = latent.shape\n    x = torch.from_numpy(latent.T).unsqueeze(0).float().to(device)  # (1, d, n)\n    \n    with torch.no_grad():\n        if isinstance(model, RVQVAE):\n            # For RVQVAE, use decoder directly\n            decoded = model.decoder(x)  # (1, D_POSE, N)\n        else:\n            # For regular decoder\n            decoded = model(x)  # (1, D_POSE, N)\n        motion_np = decoded.squeeze(0).cpu().numpy().T  # (N, D_POSE)\n    \n    return motion_np\n\n\ndef encode_motion_to_tokens(features, model, device='cuda' if torch.cuda.is_available() else 'cpu'):\n    \"\"\"\n    Encode motion to discrete tokens using RVQVAE.\n    \n    Args:\n        features: (N, D_POSE) numpy array\n        model: RVQVAE model\n        device: device to run on\n    \n    Returns:\n        tokens: List of (n,) numpy arrays, one per quantization layer\n    \"\"\"\n    if not isinstance(model, RVQVAE):\n        raise ValueError(\"Model must be RVQVAE for token encoding\")\n    \n    model.eval()\n    model.to(device)\n    \n    # Prepare input: (1, D_POSE, N)\n    N, D = features.shape\n    x = torch.from_numpy(features.T).unsqueeze(0).float().to(device)  # (1, D, N)\n    \n    with torch.no_grad():\n        tokens = model.encode_to_tokens(x)  # List of (1, n) tensors\n        tokens_np = [t.squeeze(0).cpu().numpy() for t in tokens]  # List of (n,) arrays\n    \n    return tokens_np\n\n\n\n\n\n\n\ndef load_rvq_vae(checkpoint_path, device='cuda' if torch.cuda.is_available() else 'cpu'):\n    \"\"\"\n    Load trained RVQVAE model from checkpoint.\n    \n    Args:\n        checkpoint_path: Path to RVQVAE checkpoint (.pth file)\n        device: Device to load model on\n    \n    Returns:\n        model: RVQVAE model\n        config: Dictionary with model configuration\n    \"\"\"\n    checkpoint = torch.load(checkpoint_path, map_location=device)\n    \n    # Extract config from checkpoint\n    if 'config' in checkpoint:\n        config = checkpoint['config']\n    else:\n        print(\"Warning: No config found in checkpoint, using defaults\")\n        config = {}\n    \n    # Get model parameters from config or use defaults\n    input_dim = config.get('input_dim', config.get('D_POSE', 256))\n    output_dim = config.get('output_dim', config.get('D_POSE', 256))\n    latent_dim = config.get('latent_dim', 256)\n    downsampling_ratio = config.get('downsampling_ratio', 4)\n    num_layers = config.get('num_layers', 4)\n    hidden_dim = config.get('hidden_dim', 512)\n    num_quantizers = config.get('num_quantizers', 6)\n    num_embeddings = config.get('num_embeddings', 512)\n    commitment_cost = config.get('commitment_cost', 1.0)\n    decay = config.get('decay', 0.99)\n    quantization_dropout = config.get('quantization_dropout', 0.2)\n    \n    # Create model\n    model = RVQVAE(\n        input_dim=input_dim,\n        output_dim=output_dim,\n        latent_dim=latent_dim,\n        downsampling_ratio=downsampling_ratio,\n        num_layers=num_layers,\n        hidden_dim=hidden_dim,\n        num_quantizers=num_quantizers,\n        num_embeddings=num_embeddings,\n        commitment_cost=commitment_cost,\n        decay=decay,\n        quantization_dropout=quantization_dropout\n    )\n    \n    # Load state dict\n    if 'model_state_dict' in checkpoint:\n        model.load_state_dict(checkpoint['model_state_dict'])\n    elif 'state_dict' in checkpoint:\n        model.load_state_dict(checkpoint['state_dict'])\n    else:\n        model.load_state_dict(checkpoint)\n    \n    model.to(device)\n    model.eval()\n    \n    return model, config","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:55:26.059158Z","iopub.execute_input":"2026-04-16T13:55:26.059561Z","iopub.status.idle":"2026-04-16T13:55:26.116067Z","shell.execute_reply.started":"2026-04-16T13:55:26.059520Z","shell.execute_reply":"2026-04-16T13:55:26.114898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def convert_submission_to_npy(submission_df, vae_model, output_dir=\"predicted_motions\", device='cuda'):\n    os.makedirs(output_dir, exist_ok=True)\n    \n    # Load normalization data (Update path to wherever your normalization.npz is!)\n    norm_path = \"/kaggle/working/normalization.npz\"\n    if os.path.exists(norm_path):\n        norm_data = np.load(norm_path)\n        mean = torch.tensor(norm_data['mean'], device=device).float().unsqueeze(-1)\n        std = torch.tensor(norm_data['std'], device=device).float().unsqueeze(-1)\n    else:\n        print(\"⚠️ WARNING: normalization.npz not found! Output will NOT be denormalized.\")\n        mean, std = 0.0, 1.0\n\n    vae_model.eval()\n    \n    print(f\"Decoding {len(submission_df)} sequences to .npy...\")\n    for _, row in tqdm.tqdm(submission_df.iterrows(), total=len(submission_df)):\n        sid = row['id']\n        \n        # 1. Extract the 6 layers of tokens from the dataframe\n        layer_names = ['base_tokens', 'residual_1', 'residual_2', 'residual_3', 'residual_4', 'residual_5']\n        \n        token_layers = []\n        for col in layer_names:\n            # Convert string \"45 12 511\" -> tensor([[45, 12, 511]])\n            tokens = [int(x) for x in row[col].split()]\n            token_tensor = torch.tensor([tokens], dtype=torch.long, device=device)\n            token_layers.append(token_tensor)\n            \n        # 2. Decode using the author's function\n        with torch.no_grad():\n            # decode_from_tokens expects a list of token tensors\n            motion = vae_model.decode_from_tokens(token_layers)\n            \n            # 3. Denormalize\n            # motion is (Batch=1, Features=668, Time=N)\n            # mean/std must broadcast to (1, 668, 1)\n            motion = motion * std + mean\n            \n            # 4. Convert to Numpy and transpose to (Time, Features) for standard motion files\n            motion_np = motion.squeeze(0).permute(1, 0).cpu().numpy()\n            \n        # 5. Save to disk\n        save_path = os.path.join(output_dir, f\"{sid}.npy\")\n        np.save(save_path, motion_np)\n        \n    print(f\"🎉 Successfully saved {len(submission_df)} .npy files to '{output_dir}/'\")\n\n# --- Execute it ---\n# convert_submission_to_npy(submission, vae_model, output_dir=\"/kaggle/working/final_npy_outputs\", device=device)\n\n# Load the model using the author's built-in function\nvae_model, vae_config = load_rvq_vae(\n    \"/kaggle/input/models/antonygithinji/motion-s-vae-rvq/pytorch/default/3/rvq_vae_best.pth\", \n    device=device\n)\nconvert_submission_to_npy(submission_head, vae_model, output_dir=\"/kaggle/working/final_npy_outputs\", device=device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:55:26.119594Z","iopub.execute_input":"2026-04-16T13:55:26.119974Z","iopub.status.idle":"2026-04-16T13:55:27.620065Z","shell.execute_reply.started":"2026-04-16T13:55:26.119944Z","shell.execute_reply":"2026-04-16T13:55:27.619171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip uninstall kiseki -y\n!pip install git+https://github.com/1997MarsRover/kiseki.git","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:55:27.621269Z","iopub.execute_input":"2026-04-16T13:55:27.621653Z","iopub.status.idle":"2026-04-16T13:55:44.331361Z","shell.execute_reply.started":"2026-04-16T13:55:27.621613Z","shell.execute_reply":"2026-04-16T13:55:44.330171Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install gdown \n!pip install gdown","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:55:44.333306Z","iopub.execute_input":"2026-04-16T13:55:44.333660Z","iopub.status.idle":"2026-04-16T13:55:48.653166Z","shell.execute_reply.started":"2026-04-16T13:55:44.333621Z","shell.execute_reply":"2026-04-16T13:55:48.651427Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!gdown --folder https://drive.google.com/drive/folders/1UygeyjTqH8uIuv9MhvJuvBEmlshz0r0W","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:55:48.654895Z","iopub.execute_input":"2026-04-16T13:55:48.655309Z","iopub.status.idle":"2026-04-16T13:56:11.801035Z","shell.execute_reply.started":"2026-04-16T13:55:48.655268Z","shell.execute_reply":"2026-04-16T13:56:11.799748Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport pandas as pd\nimport numpy as np\nimport torch\nfrom tqdm import tqdm\n\n# ==========================================\n# 1. CONFIGURATION PATHS\n# ==========================================\nDATA_DIR = '/kaggle/input/datasets/maksimrazantsau/asl-dataset/dataset_val_asl'  # Your unzipped ASL dataset\nOUTPUT_CSV = '/kaggle/working/new_asl_tokens.csv'\nNORM_PATH = \"/kaggle/working/normalization.npz\" # Ensure this points to your actual norm file\nVAE_CKPT = \"/kaggle/input/models/antonygithinji/motion-s-vae-rvq/pytorch/default/3/rvq_vae_best.pth\"\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\n# ==========================================\n# 2. LOAD DEPENDENCIES (Norm & VAE)\n# ==========================================\nprint(\"Loading Normalization Stats...\")\nif os.path.exists(NORM_PATH):\n    norm_data = np.load(NORM_PATH)\n    # Load as numpy arrays for fast CPU-side preprocessing\n    mean = norm_data['mean']\n    std = norm_data['std']\n    # Safety feature: Prevent division by zero if std is perfectly 0 anywhere\n    std = np.where(std == 0, 1e-8, std) \nelse:\n    raise FileNotFoundError(f\"❌ FATAL: Could not find {NORM_PATH}. You must have normalization data!\")\n\nprint(\"Loading RVQ-VAE...\")\nvae_model, vae_config = load_rvq_vae(VAE_CKPT, device=device)\nvae_model.eval()\n\n# ==========================================\n# 3. END-TO-END DATASET GENERATOR\n# ==========================================\ndataset_rows = []\nclip_folders = sorted(glob.glob(os.path.join(DATA_DIR, '*')))\n\nprint(f\"\\n🚀 Tokenizing {len(clip_folders)} ASL folders...\")\n\nfor folder in tqdm(clip_folders, desc=\"Processing ASL Dataset\"):\n    clip_id = os.path.basename(folder)\n    \n    # Locate files (Adjust 'features.npy' if your npy files are named differently, e.g. f\"{clip_id}.npy\")\n    npy_path = os.path.join(folder, 'features.npy') \n    txt_path = os.path.join(folder, 'text.txt')\n    \n    if not os.path.exists(npy_path) or not os.path.exists(txt_path):\n        # Skip if folder is missing files\n        continue\n        \n    # --- A: Read and Tag Text ---\n    with open(txt_path, 'r', encoding='utf-8') as f:\n        raw_text = f.read().strip()\n    \n    tagged_sentence = f\"[ASL] {raw_text}\"\n    \n    # --- B: Load and Normalize Motion ---\n    motion_npy = np.load(npy_path)\n    # Scale it down so the VAE can read it\n    motion_npy_norm = (motion_npy - mean) / std \n    \n    # --- C: Encode to Tokens ---\n    try:\n        # Bypassing the transformer to get raw ground-truth tokens from your VAE\n        tokens_list = encode_motion_to_tokens(motion_npy_norm, vae_model, device)\n    except Exception as e:\n        print(f\"\\n⚠️ Error tokenizing {clip_id}: {e}\")\n        continue\n        \n    # Format the lists of integers into space-separated strings\n    token_strings = [\" \".join(map(str, layer)) for layer in tokens_list]\n    \n    # Pad with empty strings if your VAE happens to have fewer than 6 quantizers\n    while len(token_strings) < 6:\n        token_strings.append(\"\")\n    \n    # --- D: Pack it for the Training Script ---\n    dataset_rows.append({\n        'id': clip_id,\n        'sentence': tagged_sentence,\n        'gloss': tagged_sentence, # Duplicated to satisfy your custom training DataLoader\n        'base_tokens': token_strings[0],\n        'residual_1': token_strings[1],\n        'residual_2': token_strings[2],\n        'residual_3': token_strings[3],\n        'residual_4': token_strings[4],\n        'residual_5': token_strings[5]\n    })\n\n# ==========================================\n# 4. SAVE FINAL DATASET\n# ==========================================\nfinal_df = pd.DataFrame(dataset_rows)\nfinal_df.to_csv(OUTPUT_CSV, index=False)\n\nprint(\"\\n\" + \"=\"*60)\nprint(f\"🎉 SUCCESS! Dataset generated with {len(final_df)} validated items.\")\nprint(f\"Saved directly to: {OUTPUT_CSV}\")\nprint(\"=\"*60)\n\n# Quick sanity check preview\nprint(\"\\nPreview of formatting:\")\nprint(final_df[['id', 'sentence', 'base_tokens']].head(2))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:56:11.802561Z","iopub.execute_input":"2026-04-16T13:56:11.802900Z","iopub.status.idle":"2026-04-16T13:56:48.205323Z","shell.execute_reply.started":"2026-04-16T13:56:11.802829Z","shell.execute_reply":"2026-04-16T13:56:48.204355Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\n# 1. Load a single file from YOUR new dataset (Update folder name if needed)\nmy_motion = np.load('/kaggle/input/datasets/maksimrazantsau/asl-dataset/dataset_val_asl/-d5dN54tH2E_11-1-rgb_front/features.npy')\n\n# 2. Load the REAL competition file you used earlier\ncomp_motion = np.load('/kaggle/working/final_npy_outputs/6420249.npy')\n\nprint(\"=== SHAPE CHECK ===\")\nprint(f\"My Motion Shape: {my_motion.shape}\")\nprint(f\"Comp Motion Shape: {comp_motion.shape}\")\n\nprint(\"\\n=== ROOT HEIGHT (Y-Axis Translation) ===\")\n# Root Y is index 3 in the 668 array\nprint(f\"My Pelvis Height:   {my_motion[:, 3].mean():.4f} meters\")\nprint(f\"Comp Pelvis Height: {comp_motion[:, 3].mean():.4f} meters\")\n\nprint(\"\\n=== VELOCITIES (FPS check) ===\")\n# Local velocities are roughly indices 499 to 664\nprint(f\"My Max Velocity:   {np.abs(my_motion[:, 499:664]).max():.6f}\")\nprint(f\"Comp Max Velocity: {np.abs(comp_motion[:, 499:664]).max():.6f}\")\n\nprint(\"\\n=== NORMALIZATION BLOWOUT CHECK ===\")\n# Let's see what the VAE actually sees after normalization!\nnorm_data = np.load(\"/kaggle/working/normalization.npz\") # Use your actual norm path\nmean = norm_data['mean']\nstd = np.where(norm_data['std'] == 0, 1e-8, norm_data['std'])\n\nmy_norm = (my_motion - mean) / std\ncomp_norm = (comp_motion - mean) / std\n\nprint(f\"My Normalized Data Range:   [{my_norm.min():.2f} to {my_norm.max():.2f}]\")\nprint(f\"Comp Normalized Data Range: [{comp_norm.min():.2f} to {comp_norm.max():.2f}]\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:56:48.206473Z","iopub.execute_input":"2026-04-16T13:56:48.206740Z","iopub.status.idle":"2026-04-16T13:56:48.222242Z","shell.execute_reply.started":"2026-04-16T13:56:48.206715Z","shell.execute_reply":"2026-04-16T13:56:48.221246Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from kiseki import visualize\n\n# Focus on hands with front view\nvisualize(\"/kaggle/working/dataset_val_asl/_2FBDaOPYig_5-5-rgb_front/features.npy\",\n          focus_joints='both_hands',\n          fixed_view='front',\n          fps=30, display=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:56:48.223559Z","iopub.execute_input":"2026-04-16T13:56:48.224010Z","iopub.status.idle":"2026-04-16T13:56:48.276755Z","shell.execute_reply.started":"2026-04-16T13:56:48.223970Z","shell.execute_reply":"2026-04-16T13:56:48.275507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize(\"/kaggle/input/asl-data-motionx-format/how2sign_motion_x_features(10).npy\",\n          focus_joints='both_hands',\n          fixed_view='front',\n          fps=30, display=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:56:48.277654Z","iopub.status.idle":"2026-04-16T13:56:48.278052Z","shell.execute_reply.started":"2026-04-16T13:56:48.277830Z","shell.execute_reply":"2026-04-16T13:56:48.277887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"norm_path = \"/kaggle/working/normalization.npz\"\nif os.path.exists(norm_path): \n    norm_data = np.load(norm_path)\n    mean = torch.tensor(norm_data['mean'], device=device).float()\n    std = torch.tensor(norm_data['std'], device=device).float()\nelse:\n    print(\"⚠️ WARNING: normalization.npz not found! Output will NOT be denormalized.\")\n\n# --- a REAL motion from your dataset\nreal_motion_npy = np.load(\"/kaggle/input/competitions/motion-s-hierarchical-text-to-motion-generation-for-sign-language/Motion-Features/1000648.npy\")\nreal_motion_tensor = torch.tensor(real_motion_npy, device=device).float()\nreal_motion_npy_norm = ((real_motion_tensor - mean) / std).cpu().numpy()\n# 2. Encode it into tokens (bypassing your Transformer entirely)\nreal_tokens = encode_motion_to_tokens(real_motion_npy_norm, vae_model, device)\n\n# 3. Decode those tokens back into motion\nwith torch.no_grad():\n    token_tensors = [torch.tensor(t, device=device).unsqueeze(0) for t in real_tokens]\n    reconstructed_motion = vae_model.decode_from_tokens(token_tensors)\n    \n    # Denormalize it exactly as you do in your submission script!\n    reconstructed_motion = reconstructed_motion * std.view(1, -1, 1) + mean.view(1, -1, 1)\n    recon_np = reconstructed_motion.squeeze(0).permute(1, 0).cpu().numpy()\n\n# 4. Save and Visualize 'recon_np'\nnp.save(\"/kaggle/working/diagnostic_test_1000648.npy\", recon_np)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:56:48.279360Z","iopub.status.idle":"2026-04-16T13:56:48.279689Z","shell.execute_reply.started":"2026-04-16T13:56:48.279539Z","shell.execute_reply":"2026-04-16T13:56:48.279558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize(\"/kaggle/input/asl-data-motionx-format/how2sign_motion_x_features(1).npy\",\n          focus_joints='both_hands',\n          fixed_view='front',\n          fps=30, display=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:56:48.281529Z","iopub.status.idle":"2026-04-16T13:56:48.281958Z","shell.execute_reply.started":"2026-04-16T13:56:48.281724Z","shell.execute_reply":"2026-04-16T13:56:48.281749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from kiseki import compare\n\n\nm_a = \"/kaggle/input/competitions/motion-s-hierarchical-text-to-motion-generation-for-sign-language/Motion-Features/1000648.npy\"\nm_b = \"/kaggle/working/diagnostic_test_1000648.npy\"\n# Overlay -- both skeletons on the same axes\ncompare(m_a,\n        m_b, \n        mode=\"overlay\", display=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-16T13:56:48.283286Z","iopub.status.idle":"2026-04-16T13:56:48.283681Z","shell.execute_reply.started":"2026-04-16T13:56:48.283509Z","shell.execute_reply":"2026-04-16T13:56:48.283530Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}