{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.18","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":46105,"databundleVersionId":5087314,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-07-10T01:26:17.005062Z","iopub.execute_input":"2025-07-10T01:26:17.007147Z","iopub.status.idle":"2025-07-10T01:26:17.457739Z","shell.execute_reply.started":"2025-07-10T01:26:17.007116Z","shell.execute_reply":"2025-07-10T01:26:17.453023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv = pd.read_csv(\"/kaggle/input/asl-signs/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T01:26:17.458933Z","iopub.execute_input":"2025-07-10T01:26:17.459252Z","iopub.status.idle":"2025-07-10T01:26:17.630704Z","shell.execute_reply.started":"2025-07-10T01:26:17.459227Z","shell.execute_reply":"2025-07-10T01:26:17.624966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T01:26:17.633019Z","iopub.execute_input":"2025-07-10T01:26:17.633254Z","iopub.status.idle":"2025-07-10T01:26:17.665556Z","shell.execute_reply.started":"2025-07-10T01:26:17.633233Z","shell.execute_reply":"2025-07-10T01:26:17.659349Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"critical_emergency_signs = [\n    \"bad\", \"boy\", \"child\", \"dad\", \"fireman\", \"girl\", \"go\", \"home\", \n    \"hot\", \"listen\", \"look\", \"man\", \"mom\", \"no\", \"outside\", \"owie\", \n    \"person\", \"police\", \"sick\", \"there\", \"wait\", \"water\", \"where\", \"yes\",\n    \"airplane\", \"all\", \"arm\", \"black\", \"blow\", \"blue\", \"boat\", \"car\", \n    \"close\", \"cry\", \"cut\", \"down\", \"dry\", \"ear\", \"eye\", \"face\", \"fall\", \n    \"fast\", \"feet\", \"finger\", \"green\", \"happy\", \"head\", \"hear\", \n    \"helicopter\", \"later\", \"loud\", \"mad\", \"many\", \"mouth\", \"nose\", \n    \"now\", \"open\", \"quiet\", \"red\", \"sad\", \"say\", \"see\", \"talk\", \n    \"touch\", \"up\", \"wet\", \"white\", \"who\", \"why\", \"yellow\", \"alligator\", \"animal\", \"backyard\", \"bed\", \"because\", \"after\", \"another\", \"any\", \"bedroom\", \"bee\", \"before\", \"beside\", \"bug\", \"can\", \"cat\", \"cheek\", \"chin\", \"dog\", \"drop\", \"find\", \n    \"for\", \"give\", \"glasswindow\", \"grandma\", \"grandpa\", \"hair\", \"have\", \"haveto\", \"hesheit\", \"jump\",\"if\", \n    \"into\", \"hide\", \"high\",\"like\",  \"night\", \"noisy\", \n    \"not\", \"please\", \"pool\", \"room\", \"stairs\", \"stuck\", \"think\", \"that\", \"time\", \"tree\", \"will\", \n]\n\n# filter out critical emergency signs\ndf_filtered = train_csv[train_csv['sign'].isin(critical_emergency_signs)]\ndf_filtered","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T01:26:17.667139Z","iopub.execute_input":"2025-07-10T01:26:17.667550Z","iopub.status.idle":"2025-07-10T01:26:17.695429Z","shell.execute_reply.started":"2025-07-10T01:26:17.667527Z","shell.execute_reply":"2025-07-10T01:26:17.690353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_filtered.loc[df_filtered[\"participant_id\"] == 32319]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T01:26:17.696986Z","iopub.execute_input":"2025-07-10T01:26:17.697407Z","iopub.status.idle":"2025-07-10T01:26:17.715662Z","shell.execute_reply.started":"2025-07-10T01:26:17.697385Z","shell.execute_reply":"2025-07-10T01:26:17.709886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"parquet_files = df_filtered.loc[df_filtered[\"participant_id\"] == 32319][\"path\"].to_list()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T01:26:17.717787Z","iopub.execute_input":"2025-07-10T01:26:17.717994Z","iopub.status.idle":"2025-07-10T01:26:17.729362Z","shell.execute_reply.started":"2025-07-10T01:26:17.717975Z","shell.execute_reply":"2025-07-10T01:26:17.723090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"folder_path = \"/kaggle/input/asl-signs/\"\n\n# Read, label with path, and concatenate\ndf_parquet = pd.concat(\n    [\n        pd.read_parquet(os.path.join(folder_path, path)).assign(path=path)\n        for path in parquet_files\n    ],\n    ignore_index=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T01:26:17.731263Z","iopub.execute_input":"2025-07-10T01:26:17.731465Z","iopub.status.idle":"2025-07-10T01:27:06.938295Z","shell.execute_reply.started":"2025-07-10T01:26:17.731446Z","shell.execute_reply":"2025-07-10T01:27:06.933038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_parquet.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T01:27:06.940861Z","iopub.execute_input":"2025-07-10T01:27:06.941216Z","iopub.status.idle":"2025-07-10T01:27:06.961951Z","shell.execute_reply.started":"2025-07-10T01:27:06.941156Z","shell.execute_reply":"2025-07-10T01:27:06.955866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"merged_df = df_parquet.merge(df_filtered, on=\"path\", how=\"left\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T01:27:06.964889Z","iopub.execute_input":"2025-07-10T01:27:06.965128Z","iopub.status.idle":"2025-07-10T01:27:20.612167Z","shell.execute_reply.started":"2025-07-10T01:27:06.965106Z","shell.execute_reply":"2025-07-10T01:27:20.606403Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"merged_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T01:27:20.613890Z","iopub.execute_input":"2025-07-10T01:27:20.614111Z","iopub.status.idle":"2025-07-10T01:27:20.698281Z","shell.execute_reply.started":"2025-07-10T01:27:20.614090Z","shell.execute_reply":"2025-07-10T01:27:20.691205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# merged_df.drop(columns=\"participant_id\").to_csv(\"train_landmark_files_2018.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T01:27:20.700046Z","iopub.execute_input":"2025-07-10T01:27:20.700292Z","iopub.status.idle":"2025-07-10T01:27:20.710934Z","shell.execute_reply.started":"2025-07-10T01:27:20.700269Z","shell.execute_reply":"2025-07-10T01:27:20.705627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"merged_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T01:27:20.713289Z","iopub.execute_input":"2025-07-10T01:27:20.713496Z","iopub.status.idle":"2025-07-10T01:27:20.726025Z","shell.execute_reply.started":"2025-07-10T01:27:20.713478Z","shell.execute_reply":"2025-07-10T01:27:20.721845Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport warnings\n\n# Suppress TensorFlow warnings and GPU messages\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '2'  # Suppress INFO and WARNING logs\nos.environ['CUDA_VISIBLE_DEVICES'] = '0,1'  # Explicitly set GPU devices for T4 x2\n\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nfrom typing import Dict, List, Tuple, Optional\nfrom sklearn.model_selection import train_test_split\n\n# Additional GPU configuration for T4 x2\ndef configure_gpu():\n    \"\"\"Configure GPU settings for optimal performance on T4 x2\"\"\"\n    gpus = tf.config.experimental.list_physical_devices('GPU')\n    if gpus:\n        try:\n            # Enable memory growth to avoid OOM errors\n            for gpu in gpus:\n                tf.config.experimental.set_memory_growth(gpu, True)\n            \n            # Set up multi-GPU strategy for T4 x2\n            strategy = tf.distribute.MirroredStrategy()\n            print(f\"Number of replicas: {strategy.num_replicas_in_sync}\")\n            return strategy\n        except RuntimeError as e:\n            print(f\"GPU configuration error: {e}\")\n            return None\n    else:\n        print(\"No GPUs found, using CPU\")\n        return None\n\n# Configure GPU before importing the rest\nstrategy = configure_gpu()\n\n# Constants\nROWS_PER_FRAME = 543\nMAX_LEN = 384\nPAD = -100.0\n\nNOSE = [1, 2, 98, 327]\nLNOSE = [98]\nRNOSE = [327]\nLIP = [0, 61, 185, 40, 39, 37, 267, 269, 270, 409, 291, 146, 91, 181, 84, 17, 314, 405, 321, 375, 78, 191, 80, 81, 82, 13, 312, 311, 310, 415, 95, 88, 178, 87, 14, 317, 402, 318, 324, 308]\nLLIP = [84, 181, 91, 146, 61, 185, 40, 39, 37, 87, 178, 88, 95, 78, 191, 80, 81, 82]\nRLIP = [314, 405, 321, 375, 291, 409, 270, 269, 267, 317, 402, 318, 324, 308, 415, 310, 311, 312]\nPOSE = [500, 502, 504, 501, 503, 505, 512, 513]\nLPOSE = [513, 505, 503, 501]\nRPOSE = [512, 504, 502, 500]\nREYE = [33, 7, 163, 144, 145, 153, 154, 155, 133, 246, 161, 160, 159, 158, 157, 173]\nLEYE = [263, 249, 390, 373, 374, 380, 381, 382, 362, 466, 388, 387, 386, 385, 384, 398]\nLHAND = list(range(468, 489))\nRHAND = list(range(522, 543))\nPOINT_LANDMARKS = LIP + LHAND + RHAND + NOSE + REYE + LEYE\nNUM_NODES = len(POINT_LANDMARKS)\nCHANNELS = 6 * NUM_NODES\n\nclass SignLanguageProcessor:\n    def __init__(self):\n        self.sign_to_index = {}\n        self.index_to_sign = {}\n        self.sign_count = 0\n        \n    def analyze_data_structure(self, df: pd.DataFrame):\n        \"\"\"Analyze the structure of the input data to understand format\"\"\"\n        print(\"=== Data Structure Analysis ===\")\n        print(f\"Columns: {df.columns.tolist()}\")\n        print(f\"Shape: {df.shape}\")\n        \n        # Analyze row_id format\n        sample_row_ids = df['row_id'].head(20).tolist()\n        print(f\"Sample row_ids: {sample_row_ids}\")\n        \n        # Check if row_ids are numeric or string-based\n        numeric_count = 0\n        string_count = 0\n        \n        for row_id in sample_row_ids:\n            try:\n                int(row_id)\n                numeric_count += 1\n            except (ValueError, TypeError):\n                string_count += 1\n        \n        print(f\"Numeric row_ids: {numeric_count}, String-based row_ids: {string_count}\")\n        \n        # Analyze frame structure\n        print(f\"Frame range: {df['frame'].min()} to {df['frame'].max()}\")\n        print(f\"Total frames: {df['frame'].nunique()}\")\n        print(f\"Unique signs: {df['sign'].nunique()}\")\n        print(f\"Sample signs: {df['sign'].unique()[:10]}\")\n        \n        # Check for missing values\n        print(f\"Missing values per column:\")\n        for col in ['x', 'y', 'z']:\n            missing = df[col].isna().sum()\n            print(f\"  {col}: {missing} ({missing/len(df)*100:.1f}%)\")\n        \n        # Additional debugging info\n        print(f\"Unique row_id count: {df['row_id'].nunique()}\")\n        print(f\"Records per frame (avg): {len(df) / df['frame'].nunique():.1f}\")\n        \n        print(\"=\" * 30)\n        \n        return {\n            'has_string_row_ids': string_count > numeric_count,\n            'frame_range': (df['frame'].min(), df['frame'].max()),\n            'num_signs': df['sign'].nunique(),\n            'missing_coords': {col: df[col].isna().sum() for col in ['x', 'y', 'z']}\n        }\n    def build_sign_vocabulary(self, df: pd.DataFrame) -> List[str]:\n        unique_signs = df['sign'].unique()\n        for idx, sign in enumerate(unique_signs):\n            self.sign_to_index[sign] = idx\n            self.index_to_sign[idx] = sign\n        self.sign_count = len(unique_signs)\n        return unique_signs.tolist()\n    \n    def group_by_sequence(self, df: pd.DataFrame) -> Dict:\n        print(\"\\n=== Grouping by Sequence ===\")\n        # Check actual column names and adapt\n        if 'participant_id' in df.columns and 'sequence_id' in df.columns:\n            print(\"Using participant_id + sequence_id for grouping\")\n            df['seq_key'] = df['participant_id'].astype(str) + '_' + df['sequence_id'].astype(str)\n        elif 'path' in df.columns:\n            print(\"Using path column for grouping\")\n            # Extract participant and sequence from path\n            df['seq_key'] = df['path'].str.replace('.parquet', '').str.replace('train_landmark_files/', '')\n        else:\n            print(\"Using frame-based grouping (fallback)\")\n            # Fallback: use frame groups if available\n            df['seq_key'] = df.index // 1000  # Rough grouping\n        \n        grouped = df.groupby('seq_key')\n        print(f\"Created {len(grouped)} sequences\")\n        \n        # Debug: Show sample sequence info\n        sample_keys = list(grouped.groups.keys())[:5]\n        print(f\"Sample sequence keys: {sample_keys}\")\n        \n        for key in sample_keys[:2]:  # Show details for first 2 sequences\n            seq_data = grouped.get_group(key)\n            print(f\"  Sequence '{key}': {len(seq_data)} records, frames {seq_data['frame'].min()}-{seq_data['frame'].max()}, sign: {seq_data['sign'].iloc[0]}\")\n        \n        return grouped\n    \n    def create_landmark_mapping(self, df: pd.DataFrame):\n        \"\"\"Create a mapping from string-based row_ids to numeric indices\"\"\"\n        print(\"\\n=== Creating Landmark Mapping ===\")\n        unique_row_ids = df['row_id'].unique()\n        print(f\"Found {len(unique_row_ids)} unique row_ids\")\n        \n        # Create mapping dictionary\n        self.row_id_mapping = {}\n        \n        # Sort unique row_ids to ensure consistent mapping\n        sorted_row_ids = sorted(unique_row_ids, key=str)\n        \n        for idx, row_id in enumerate(sorted_row_ids):\n            if idx < ROWS_PER_FRAME:  # Only map if within our expected range\n                self.row_id_mapping[row_id] = idx\n        \n        print(f\"Created mapping for {len(self.row_id_mapping)} landmark types\")\n        print(f\"Sample mapping: {dict(list(self.row_id_mapping.items())[:5])}\")\n        \n        # Show mapping statistics\n        if len(unique_row_ids) > ROWS_PER_FRAME:\n            print(f\"WARNING: {len(unique_row_ids)} unique row_ids found, but only {ROWS_PER_FRAME} slots available\")\n            print(f\"Unmapped row_ids: {len(unique_row_ids) - len(self.row_id_mapping)}\")\n        \n        return self.row_id_mapping\n    \n    def reshape_to_frames_robust(self, sequence_df: pd.DataFrame) -> np.ndarray:\n        \"\"\"Robust version that handles different row_id formats\"\"\"\n        print(f\"\\n--- Processing sequence with {len(sequence_df)} records ---\")\n        frames = []\n        \n        # Create mapping if not exists\n        if not hasattr(self, 'row_id_mapping'):\n            self.create_landmark_mapping(sequence_df)\n        \n        frame_numbers = sorted(sequence_df['frame'].unique())\n        print(f\"Processing {len(frame_numbers)} frames: {frame_numbers[:5]}{'...' if len(frame_numbers) > 5 else ''}\")\n        \n        successful_mappings = 0\n        failed_mappings = 0\n        \n        for frame_num in frame_numbers:\n            frame_data = sequence_df[sequence_df['frame'] == frame_num]\n            coordinates = np.full((ROWS_PER_FRAME, 3), np.nan)\n            \n            frame_successful = 0\n            \n            for _, row in frame_data.iterrows():\n                row_id = row['row_id']\n                \n                # Try different approaches to get index\n                idx = None\n                \n                # Approach 1: Use mapping if available\n                if hasattr(self, 'row_id_mapping') and row_id in self.row_id_mapping:\n                    idx = self.row_id_mapping[row_id]\n                \n                # Approach 2: Try parsing as number\n                elif isinstance(row_id, (int, float)):\n                    idx = int(row_id)\n                \n                # Approach 3: Extract number from string\n                else:\n                    try:\n                        idx = self.parse_row_id(row_id)\n                    except:\n                        failed_mappings += 1\n                        continue\n                \n                # Store coordinates if valid index\n                if idx is not None and 0 <= idx < ROWS_PER_FRAME:\n                    coordinates[idx] = [row['x'], row['y'], row['z']]\n                    successful_mappings += 1\n                    frame_successful += 1\n                else:\n                    failed_mappings += 1\n            \n            frames.append(coordinates)\n            \n            # Debug info for first few frames\n            if len(frames) <= 3:\n                non_nan_count = np.sum(~np.isnan(coordinates[:, 0]))\n                print(f\"  Frame {frame_num}: {frame_successful} landmarks mapped, {non_nan_count} non-NaN coordinates\")\n        \n        print(f\"Total mapping results: {successful_mappings} successful, {failed_mappings} failed\")\n        result = np.array(frames)\n        print(f\"Final frames shape: {result.shape}\")\n        \n        return result\n    def parse_row_id(self, row_id):\n        \"\"\"Parse row_id which can be either integer or string format like '18-face-0'\"\"\"\n        if isinstance(row_id, (int, float)):\n            return int(row_id)\n        \n        # Handle string format like '18-face-0', 'left_hand_21', etc.\n        if isinstance(row_id, str):\n            # Try to extract the last number after the last dash or underscore\n            import re\n            numbers = re.findall(r'\\d+', row_id)\n            if numbers:\n                # Use the last number found\n                return int(numbers[-1])\n            else:\n                # If no numbers found, try to map landmark types to indices\n                return self.map_landmark_type_to_index(row_id)\n        \n        return 0  # fallback\n    \n    def map_landmark_type_to_index(self, row_id):\n        \"\"\"Map landmark type strings to indices\"\"\"\n        # Create a mapping for different landmark types\n        landmark_mapping = {}\n        \n        # Face landmarks (0-467)\n        if 'face' in row_id.lower():\n            return 0  # Start of face landmarks\n        \n        # Left hand landmarks (468-488)  \n        elif 'left_hand' in row_id.lower():\n            return 468\n        \n        # Right hand landmarks (489-509)\n        elif 'right_hand' in row_id.lower():\n            return 489\n        \n        # Pose landmarks (510-542)\n        elif 'pose' in row_id.lower():\n            return 510\n        \n        return 0  # fallback\n    \n    def reshape_to_frames(self, sequence_df: pd.DataFrame) -> np.ndarray:\n        \"\"\"Use the robust version by default\"\"\"\n        return self.reshape_to_frames_robust(sequence_df)\n    \n    def filter_nans(self, frames: np.ndarray) -> np.ndarray:\n        print(f\"\\n--- Filtering NaN frames ---\")\n        print(f\"Input frames shape: {frames.shape}\")\n        \n        valid_frames = []\n        for i, frame in enumerate(frames):\n            ref_points = frame[POINT_LANDMARKS]\n            if not np.all(np.isnan(ref_points[:, :2])):\n                valid_frames.append(frame)\n            elif i < 5:  # Debug first few frames\n                nan_count = np.sum(np.isnan(ref_points[:, :2]))\n                print(f\"  Frame {i}: {nan_count}/{len(ref_points)*2} NaN values in reference points\")\n        \n        result = np.array(valid_frames) if valid_frames else np.array([])\n        print(f\"Filtered from {len(frames)} to {len(result)} valid frames\")\n        \n        return result\n    \n    def tf_nan_mean(self, x, axis=0, keepdims=False):\n        return tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), x), axis=axis, keepdims=keepdims) / \\\n               tf.reduce_sum(tf.where(tf.math.is_nan(x), tf.zeros_like(x), tf.ones_like(x)), axis=axis, keepdims=keepdims)\n    \n    def tf_nan_std(self, x, center=None, axis=0, keepdims=False):\n        if center is None:\n            center = self.tf_nan_mean(x, axis=axis, keepdims=True)\n        d = x - center\n        return tf.math.sqrt(self.tf_nan_mean(d * d, axis=axis, keepdims=keepdims))\n    \n    def preprocess_sequence(self, frames: np.ndarray) -> np.ndarray:\n        print(f\"--- Preprocessing sequence with {len(frames)} frames ---\")\n        \n        if len(frames) == 0:\n            print(\"Empty frames array, returning empty result\")\n            return np.array([])\n        \n        x = tf.constant(frames, dtype=tf.float32)\n        if tf.rank(x) == 3:\n            x = x[None, ...]\n        \n        print(f\"Input tensor shape: {x.shape}\")\n        \n        mean = self.tf_nan_mean(tf.gather(x, [17], axis=2), axis=[1, 2], keepdims=True)\n        mean = tf.where(tf.math.is_nan(mean), tf.constant(0.5, x.dtype), mean)\n        \n        x = tf.gather(x, POINT_LANDMARKS, axis=2)\n        print(f\"After gathering landmarks: {x.shape}\")\n        \n        std = self.tf_nan_std(x, center=mean, axis=[1, 2], keepdims=True)\n        x = (x - mean) / std\n        \n        if x.shape[1] > MAX_LEN:\n            print(f\"Trimming sequence from {x.shape[1]} to {MAX_LEN} frames\")\n            x = x[:, :MAX_LEN]\n        \n        length = tf.shape(x)[1]\n        x = x[..., :2]\n        \n        dx = tf.cond(tf.shape(x)[1] > 1,\n                    lambda: tf.pad(x[:, 1:] - x[:, :-1], [[0, 0], [0, 1], [0, 0], [0, 0]]),\n                    lambda: tf.zeros_like(x))\n        \n        dx2 = tf.cond(tf.shape(x)[1] > 2,\n                     lambda: tf.pad(x[:, 2:] - x[:, :-2], [[0, 0], [0, 2], [0, 0], [0, 0]]),\n                     lambda: tf.zeros_like(x))\n        \n        x = tf.concat([\n            tf.reshape(x, (-1, length, 2 * len(POINT_LANDMARKS))),\n            tf.reshape(dx, (-1, length, 2 * len(POINT_LANDMARKS))),\n            tf.reshape(dx2, (-1, length, 2 * len(POINT_LANDMARKS))),\n        ], axis=-1)\n        \n        x = tf.where(tf.math.is_nan(x), tf.constant(0., x.dtype), x)\n        result = x[0].numpy()\n        print(f\"Final preprocessed shape: {result.shape}\")\n        \n        return result\n    \n    def crop_or_pad(self, sequence: np.ndarray, max_len: int = MAX_LEN) -> np.ndarray:\n        if len(sequence) >= max_len:\n            return sequence[:max_len]\n        \n        padding = np.full((max_len - len(sequence), CHANNELS), PAD)\n        return np.vstack([sequence, padding])\n    \n    def one_hot_encode(self, sign_label: str) -> np.ndarray:\n        index = self.sign_to_index.get(sign_label, 0)\n        one_hot = np.zeros(self.sign_count)\n        one_hot[index] = 1\n        return one_hot\n    \n    def process_sequence(self, sequence_df: pd.DataFrame, sign: str) -> Optional[Tuple[np.ndarray, np.ndarray]]:\n        print(f\"\\n=== Processing sequence for sign: '{sign}' ===\")\n        print(f\"Sequence data shape: {sequence_df.shape}\")\n        \n        frames = self.reshape_to_frames(sequence_df)\n        if len(frames) == 0:\n            print(\"❌ No frames generated, skipping sequence\")\n            return None\n        \n        filtered_frames = self.filter_nans(frames)\n        if len(filtered_frames) == 0:\n            print(\"❌ No valid frames after filtering, skipping sequence\")\n            return None\n        \n        processed_features = self.preprocess_sequence(filtered_frames)\n        if len(processed_features) == 0:\n            print(\"❌ No features after preprocessing, skipping sequence\")\n            return None\n        \n        final_sequence = self.crop_or_pad(processed_features)\n        one_hot_sign = self.one_hot_encode(sign)\n        \n        print(f\"✅ Successfully processed sequence: {final_sequence.shape}\")\n        \n        return final_sequence, one_hot_sign\n    \n    def process_dataset(self, df: pd.DataFrame) -> Tuple[np.ndarray, np.ndarray]:\n        print(\"\\n🚀 Starting dataset processing...\")\n        \n        # First analyze the data structure\n        structure_info = self.analyze_data_structure(df)\n        \n        print(\"\\n📝 Building sign vocabulary...\")\n        self.build_sign_vocabulary(df)\n        print(f\"Built vocabulary with {self.sign_count} signs: {list(self.sign_to_index.keys())[:10]}{'...' if self.sign_count > 10 else ''}\")\n        \n        grouped = self.group_by_sequence(df)\n        \n        X, y = [], []\n        successful_sequences = 0\n        failed_sequences = 0\n        \n        print(f\"\\n🔄 Processing {len(grouped)} sequences...\")\n        \n        for i, (seq_key, sequence_df) in enumerate(grouped):\n            if i < 5 or i % 100 == 0:  # Show progress for first 5 and every 100th\n                print(f\"\\nProcessing sequence {i+1}/{len(grouped)}: {seq_key}\")\n            \n            sign = sequence_df['sign'].iloc[0]\n            result = self.process_sequence(sequence_df, sign)\n            \n            if result is not None:\n                features, label = result\n                X.append(features)\n                y.append(label)\n                successful_sequences += 1\n            else:\n                failed_sequences += 1\n                if failed_sequences <= 5:  # Show first few failures\n                    print(f\"❌ Failed to process sequence {seq_key}\")\n        \n        final_X = np.array(X) if X else np.array([])\n        final_y = np.array(y) if y else np.array([])\n        \n        print(f\"\\n📊 Dataset processing complete!\")\n        print(f\"✅ Successfully processed: {successful_sequences} sequences\")\n        print(f\"❌ Failed to process: {failed_sequences} sequences\")\n        print(f\"📈 Success rate: {successful_sequences/(successful_sequences+failed_sequences)*100:.1f}%\")\n        \n        return final_X, final_y\n\n# Model Components\nclass ECA(tf.keras.layers.Layer):\n    def __init__(self, kernel_size=5, **kwargs):\n        super().__init__(**kwargs)\n        self.supports_masking = True\n        self.kernel_size = kernel_size\n        self.conv = tf.keras.layers.Conv1D(1, kernel_size=kernel_size, strides=1, padding=\"same\", use_bias=False)\n\n    def call(self, inputs, mask=None):\n        nn = tf.keras.layers.GlobalAveragePooling1D()(inputs, mask=mask)\n        nn = tf.expand_dims(nn, -1)\n        nn = self.conv(nn)\n        nn = tf.squeeze(nn, -1)\n        nn = tf.nn.sigmoid(nn)\n        nn = nn[:,None,:]\n        return inputs * nn\n\nclass LateDropout(tf.keras.layers.Layer):\n    def __init__(self, rate, noise_shape=None, start_step=0, **kwargs):\n        super().__init__(**kwargs)\n        self.supports_masking = True\n        self.rate = rate\n        self.start_step = start_step\n        self.dropout = tf.keras.layers.Dropout(rate, noise_shape=noise_shape)\n      \n    def build(self, input_shape):\n        super().build(input_shape)\n        agg = tf.VariableAggregation.ONLY_FIRST_REPLICA\n        self._train_counter = tf.Variable(0, dtype=\"int64\", aggregation=agg, trainable=False)\n\n    def call(self, inputs, training=False):\n        x = tf.cond(self._train_counter < self.start_step, lambda:inputs, lambda:self.dropout(inputs, training=training))\n        if training:\n            self._train_counter.assign_add(1)\n        return x\n\nclass CausalDWConv1D(tf.keras.layers.Layer):\n    def __init__(self, kernel_size=17, dilation_rate=1, use_bias=False, depthwise_initializer='glorot_uniform', name='', **kwargs):\n        super().__init__(name=name,**kwargs)\n        self.causal_pad = tf.keras.layers.ZeroPadding1D((dilation_rate*(kernel_size-1),0),name=name + '_pad')\n        self.dw_conv = tf.keras.layers.DepthwiseConv1D(\n                            kernel_size, strides=1, dilation_rate=dilation_rate, padding='valid',\n                            use_bias=use_bias, depthwise_initializer=depthwise_initializer, name=name + '_dwconv')\n        self.supports_masking = True\n        \n    def call(self, inputs):\n        x = self.causal_pad(inputs)\n        x = self.dw_conv(x)\n        return x\n\ndef Conv1DBlock(channel_size, kernel_size, dilation_rate=1, drop_rate=0.0, expand_ratio=2, activation='swish', name=None):\n    if name is None:\n        name = str(tf.keras.backend.get_uid(\"mbblock\"))\n    \n    def apply(inputs):\n        channels_in = tf.keras.backend.int_shape(inputs)[-1]\n        channels_expand = channels_in * expand_ratio\n        skip = inputs\n\n        x = tf.keras.layers.Dense(channels_expand, use_bias=True, activation=activation, name=name + '_expand_conv')(inputs)\n        x = CausalDWConv1D(kernel_size, dilation_rate=dilation_rate, use_bias=False, name=name + '_dwconv')(x)\n        x = tf.keras.layers.BatchNormalization(momentum=0.95, name=name + '_bn')(x)\n        x = ECA()(x)\n        x = tf.keras.layers.Dense(channel_size, use_bias=True, name=name + '_project_conv')(x)\n\n        if drop_rate > 0:\n            x = tf.keras.layers.Dropout(drop_rate, noise_shape=(None,1,1), name=name + '_drop')(x)\n\n        if (channels_in == channel_size):\n            x = tf.keras.layers.add([x, skip], name=name + '_add')\n        return x\n\n    return apply\n\nclass MultiHeadSelfAttention(tf.keras.layers.Layer):\n    def __init__(self, dim=256, num_heads=4, dropout=0, **kwargs):\n        super().__init__(**kwargs)\n        self.dim = dim\n        self.scale = self.dim ** -0.5\n        self.num_heads = num_heads\n        self.qkv = tf.keras.layers.Dense(3 * dim, use_bias=False)\n        self.drop1 = tf.keras.layers.Dropout(dropout)\n        self.proj = tf.keras.layers.Dense(dim, use_bias=False)\n        self.supports_masking = True\n\n    def call(self, inputs, mask=None):\n        qkv = self.qkv(inputs)\n        qkv = tf.keras.layers.Permute((2, 1, 3))(tf.keras.layers.Reshape((-1, self.num_heads, self.dim * 3 // self.num_heads))(qkv))\n        q, k, v = tf.split(qkv, [self.dim // self.num_heads] * 3, axis=-1)\n\n        attn = tf.matmul(q, k, transpose_b=True) * self.scale\n\n        if mask is not None:\n            mask = mask[:, None, None, :]\n\n        attn = tf.keras.layers.Softmax(axis=-1)(attn, mask=mask)\n        attn = self.drop1(attn)\n\n        x = attn @ v\n        x = tf.keras.layers.Reshape((-1, self.dim))(tf.keras.layers.Permute((2, 1, 3))(x))\n        x = self.proj(x)\n        return x\n\ndef TransformerBlock(dim=256, num_heads=4, expand=4, attn_dropout=0.2, drop_rate=0.2, activation='swish'):\n    def apply(inputs):\n        x = inputs\n        x = tf.keras.layers.BatchNormalization(momentum=0.95)(x)\n        x = MultiHeadSelfAttention(dim=dim,num_heads=num_heads,dropout=attn_dropout)(x)\n        x = tf.keras.layers.Dropout(drop_rate, noise_shape=(None,1,1))(x)\n        x = tf.keras.layers.Add()([inputs, x])\n        attn_out = x\n\n        x = tf.keras.layers.BatchNormalization(momentum=0.95)(x)\n        x = tf.keras.layers.Dense(dim*expand, use_bias=False, activation=activation)(x)\n        x = tf.keras.layers.Dense(dim, use_bias=False)(x)\n        x = tf.keras.layers.Dropout(drop_rate, noise_shape=(None,1,1))(x)\n        x = tf.keras.layers.Add()([attn_out, x])\n        return x\n    return apply\n\ndef get_model(max_len=384, dropout_step=0, dim=192, num_classes=None):\n    inp = tf.keras.Input((max_len, CHANNELS))\n    x = tf.keras.layers.Masking(mask_value=PAD, input_shape=(max_len, CHANNELS))(inp)\n    ksize = 17\n    x = tf.keras.layers.Dense(dim, use_bias=False, name='stem_conv')(x)\n    x = tf.keras.layers.BatchNormalization(momentum=0.95, name='stem_bn')(x)\n\n    x = Conv1DBlock(dim, ksize, drop_rate=0.2)(x)\n    x = Conv1DBlock(dim, ksize, drop_rate=0.2)(x)\n    x = Conv1DBlock(dim, ksize, drop_rate=0.2)(x)\n    x = TransformerBlock(dim, expand=2)(x)\n\n    x = Conv1DBlock(dim, ksize, drop_rate=0.2)(x)\n    x = Conv1DBlock(dim, ksize, drop_rate=0.2)(x)\n    x = Conv1DBlock(dim, ksize, drop_rate=0.2)(x)\n    x = TransformerBlock(dim, expand=2)(x)\n\n    if dim == 384:\n        x = Conv1DBlock(dim, ksize, drop_rate=0.2)(x)\n        x = Conv1DBlock(dim, ksize, drop_rate=0.2)(x)\n        x = Conv1DBlock(dim, ksize, drop_rate=0.2)(x)\n        x = TransformerBlock(dim, expand=2)(x)\n\n        x = Conv1DBlock(dim, ksize, drop_rate=0.2)(x)\n        x = Conv1DBlock(dim, ksize, drop_rate=0.2)(x)\n        x = Conv1DBlock(dim, ksize, drop_rate=0.2)(x)\n        x = TransformerBlock(dim, expand=2)(x)\n\n    x = tf.keras.layers.Dense(dim*2, activation=None, name='top_conv')(x)\n    x = tf.keras.layers.GlobalAveragePooling1D()(x)\n    x = LateDropout(0.8, start_step=dropout_step)(x)\n    x = tf.keras.layers.Dense(num_classes, name='classifier')(x)\n    return tf.keras.Model(inp, x)\n\nclass CFG:\n    n_splits = 5\n    save_output = True\n    output_dir = './models'\n    \n    seed = 42\n    verbose = 1  # Reduced verbosity to minimize output\n    \n    max_len = 384\n    replicas = 2  # Updated for T4 x2\n    lr = 5e-4 * replicas\n    weight_decay = 0.1\n    lr_min = 1e-6\n    epoch = 50\n    warmup = 0\n    batch_size = 32 * replicas  # Increased batch size for dual GPUs\n    \n    fp16 = True\n    dropout_start_epoch = 15\n    resume = 0\n    decay_type = 'cosine'\n    dim = 192\n    comment = f'islr-fp16-192-seed{seed}-t4x2'\n\ndef create_dataset(X, y, batch_size=32, shuffle=True):\n    dataset = tf.data.Dataset.from_tensor_slices((X, y))\n    if shuffle:\n        dataset = dataset.shuffle(buffer_size=1000)\n    dataset = dataset.batch(batch_size)\n    dataset = dataset.prefetch(tf.data.AUTOTUNE)\n    return dataset\n\ndef train_model(data_input, config: CFG = None):\n    \"\"\"\n    Train the sign language model.\n    \n    Args:\n        data_input: Either a string (CSV file path) or pandas DataFrame\n        config: Configuration object\n    \"\"\"\n    if config is None:\n        config = CFG()\n    \n    os.makedirs(config.output_dir, exist_ok=True)\n    \n    # Load and check data structure\n    print(\"Loading and processing data...\")\n    \n    # Handle both DataFrame and file path inputs\n    if isinstance(data_input, pd.DataFrame):\n        df = data_input.copy()\n        print(\"Using provided DataFrame\")\n    elif isinstance(data_input, str):\n        df = pd.read_csv(data_input)\n        print(f\"Loaded CSV from: {data_input}\")\n    else:\n        raise ValueError(\"data_input must be either a pandas DataFrame or a file path string\")\n    \n    print(f\"Dataset columns: {df.columns.tolist()}\")\n    print(f\"Dataset shape: {df.shape}\")\n    \n    # Check for required columns\n    required_cols = ['frame', 'row_id', 'x', 'y', 'z', 'sign']\n    missing_cols = [col for col in required_cols if col not in df.columns]\n    if missing_cols:\n        print(f\"Missing required columns: {missing_cols}\")\n        print(\"Please ensure your CSV has these columns: frame, row_id, x, y, z, sign\")\n        return None, None, None\n    \n    processor = SignLanguageProcessor()\n    \n    # Create training function within strategy scope for multi-GPU\n    def train_step():\n        print(\"\\n🔥 Starting data processing within training scope...\")\n        X, y = processor.process_dataset(df)\n        \n        print(f\"\\n📊 Final processed dataset:\")\n        print(f\"  X shape: {X.shape}\")\n        print(f\"  y shape: {y.shape}\")\n        print(f\"  Number of classes: {processor.sign_count}\")\n        print(f\"  Samples per class: {[np.sum(y.argmax(axis=1) == i) for i in range(min(5, processor.sign_count))]}\")\n        \n        if len(X) == 0:\n            print(\"❌ No valid sequences processed! Cannot train model.\")\n            return None, None\n        \n        # Split data\n        print(f\"\\n✂️ Splitting data (80% train, 20% validation)...\")\n        X_train, X_val, y_train, y_val = train_test_split(\n            X, y, test_size=0.2, random_state=config.seed, stratify=y.argmax(axis=1)\n        )\n        \n        print(f\"  Training set: {X_train.shape}\")\n        print(f\"  Validation set: {X_val.shape}\")\n        \n        # Create datasets\n        print(f\"\\n📦 Creating TensorFlow datasets (batch size: {config.batch_size})...\")\n        train_ds = create_dataset(X_train, y_train, config.batch_size, shuffle=True)\n        val_ds = create_dataset(X_val, y_val, config.batch_size, shuffle=False)\n        \n        # Build model\n        print(f\"\\n🏗️ Building model (dim: {config.dim}, max_len: {config.max_len})...\")\n        model = get_model(\n            max_len=config.max_len,\n            dropout_step=config.dropout_start_epoch,\n            dim=config.dim,\n            num_classes=processor.sign_count\n        )\n        \n        print(f\"Model parameters: {model.count_params():,}\")\n        \n        # Compile model\n        print(f\"\\n⚙️ Compiling model (lr: {config.lr}, weight_decay: {config.weight_decay})...\")\n        optimizer = tf.keras.optimizers.AdamW(\n            learning_rate=config.lr,\n            weight_decay=config.weight_decay\n        )\n        \n        model.compile(\n            optimizer=optimizer,\n            loss=tf.keras.losses.CategoricalCrossentropy(from_logits=True),\n            metrics=['accuracy']\n        )\n        \n        # Callbacks\n        callbacks = [\n            tf.keras.callbacks.ModelCheckpoint(\n                filepath=f'{config.output_dir}/{config.comment}-best.h5',\n                monitor='val_loss',\n                save_best_only=True,\n                save_weights_only=True,\n                verbose=1\n            ),\n            tf.keras.callbacks.ReduceLROnPlateau(\n                monitor='val_loss',\n                factor=0.5,\n                patience=5,\n                min_lr=config.lr_min,\n                verbose=1\n            ),\n            tf.keras.callbacks.EarlyStopping(\n                monitor='val_loss',\n                patience=10,\n                restore_best_weights=True,\n                verbose=1\n            )\n        ]\n        \n        # Train model\n        print(f\"\\n🚀 Starting training ({config.epoch} epochs)...\")\n        print(\"=\" * 50)\n        history = model.fit(\n            train_ds,\n            epochs=config.epoch,\n            callbacks=callbacks,\n            validation_data=val_ds,\n            verbose=config.verbose\n        )\n        \n        # Load best weights\n        print(f\"\\n📥 Loading best weights...\")\n        model.load_weights(f'{config.output_dir}/{config.comment}-best.h5')\n        \n        # Evaluate\n        print(f\"\\n📈 Final evaluation...\")\n        val_loss, val_acc = model.evaluate(val_ds, verbose=0)\n        print(f\"🎯 Final validation - Loss: {val_loss:.4f}, Accuracy: {val_acc:.4f}\")\n        \n        return model, historyprint(f\"Number of classes: {processor.sign_count}\")\n        print(f\"Number of classes: {processor.sign_count}\")\n        \n        # Split data\n        X_train, X_val, y_train, y_val = train_test_split(\n            X, y, test_size=0.2, random_state=config.seed, stratify=y.argmax(axis=1)\n        )\n        \n        # Create datasets\n        train_ds = create_dataset(X_train, y_train, config.batch_size, shuffle=True)\n        val_ds = create_dataset(X_val, y_val, config.batch_size, shuffle=False)\n        \n        # Build model\n        model = get_model(\n            max_len=config.max_len,\n            dropout_step=config.dropout_start_epoch,\n            dim=config.dim,\n            num_classes=processor.sign_count\n        )\n        \n        # Compile model\n        optimizer = tf.keras.optimizers.AdamW(\n            learning_rate=config.lr,\n            weight_decay=config.weight_decay\n        )\n        \n        model.compile(\n            optimizer=optimizer,\n            loss=tf.keras.losses.CategoricalCrossentropy(from_logits=True),\n            metrics=['accuracy']\n        )\n        \n        # Callbacks\n        callbacks = [\n            tf.keras.callbacks.ModelCheckpoint(\n                filepath=f'{config.output_dir}/{config.comment}-best.h5',\n                monitor='val_loss',\n                save_best_only=True,\n                save_weights_only=True,\n                verbose=0  # Suppress checkpoint messages\n            ),\n            tf.keras.callbacks.ReduceLROnPlateau(\n                monitor='val_loss',\n                factor=0.5,\n                patience=5,\n                min_lr=config.lr_min,\n                verbose=0\n            ),\n            tf.keras.callbacks.EarlyStopping(\n                monitor='val_loss',\n                patience=10,\n                restore_best_weights=True,\n                verbose=0\n            )\n        ]\n        \n        # Train model\n        print(\"Starting training...\")\n        history = model.fit(\n            train_ds,\n            epochs=config.epoch,\n            callbacks=callbacks,\n            validation_data=val_ds,\n            verbose=config.verbose\n        )\n        \n        # Load best weights\n        model.load_weights(f'{config.output_dir}/{config.comment}-best.h5')\n        \n        # Evaluate\n        val_loss, val_acc = model.evaluate(val_ds, verbose=0)\n        print(f\"Final validation - Loss: {val_loss:.4f}, Accuracy: {val_acc:.4f}\")\n        \n        return model, history\n    \n    # Execute training within strategy scope if available\n    print(f\"\\n🎮 GPU Strategy: {'Multi-GPU' if strategy else 'Single GPU/CPU'}\")\n    \n    if strategy:\n        print(f\"Running with {strategy.num_replicas_in_sync} replicas\")\n        with strategy.scope():\n            model, history = train_step()\n    else:\n        print(\"Running without distributed strategy\")\n        model, history = train_step()\n    \n    if model is None:\n        print(\"❌ Training failed - no model returned\")\n        return None, None, None\n    \n    print(f\"\\n🎉 Training completed successfully!\")\n    return model, processor, history\n\n# Usage example with GPU optimization:\nif __name__ == \"__main__\":\n    # Print GPU information\n    print(\"=\" * 50)\n    print(\"GPU Configuration:\")\n    print(f\"TensorFlow version: {tf.__version__}\")\n    print(f\"GPUs available: {len(tf.config.experimental.list_physical_devices('GPU'))}\")\n    for i, gpu in enumerate(tf.config.experimental.list_physical_devices('GPU')):\n        print(f\"  GPU {i}: {gpu}\")\n    print(\"=\" * 50)\n    \n    # Train the model with optimized configuration\n    config = CFG()\n    \n    # Example usage with DataFrame (your case):\n    # model, processor, history = train_model(merged_df, config)\n    \n    # Example usage with CSV file:\n    # model, processor, history = train_model('train_landmark_files_2018.csv', config)\n    \n    # For demonstration, using CSV file path:\n    model, processor, history = train_model(merged_df, config)\n    \n    # Save processor for inference\n    import pickle\n    with open(f'{config.output_dir}/processor.pkl', 'wb') as f:\n        pickle.dump(processor, f)\n    \n    print(f\"Model and processor saved to {config.output_dir}/\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T01:27:20.729822Z","iopub.execute_input":"2025-07-10T01:27:20.730039Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}