{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":22422,"databundleVersionId":2153105,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install levenshtein\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T09:10:21.936283Z","iopub.execute_input":"2025-10-14T09:10:21.936494Z","iopub.status.idle":"2025-10-14T09:10:25.203098Z","shell.execute_reply.started":"2025-10-14T09:10:21.936473Z","shell.execute_reply":"2025-10-14T09:10:25.202100Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nprint(\"Num GPUs Available: \", len(tf.config.list_physical_devices('GPU')))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T09:10:25.217794Z","iopub.execute_input":"2025-10-14T09:10:25.218105Z","iopub.status.idle":"2025-10-14T09:10:29.147879Z","shell.execute_reply.started":"2025-10-14T09:10:25.218080Z","shell.execute_reply":"2025-10-14T09:10:29.147199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\nfrom sklearn.model_selection import train_test_split\nimport cv2\nfrom tqdm import tqdm\nimport warnings\nwarnings.filterwarnings('ignore')\n\nprint('Libraries imported successfully')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-14T09:10:29.148682Z","iopub.execute_input":"2025-10-14T09:10:29.149162Z","iopub.status.idle":"2025-10-14T09:10:29.209775Z","shell.execute_reply.started":"2025-10-14T09:10:29.149145Z","shell.execute_reply":"2025-10-14T09:10:29.209095Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 2: Data Loading\n# Load train_labels.csv and create image paths with nested folder structure\n\ntrain_labels = pd.read_csv('/kaggle/input/bms-molecular-translation/train_labels.csv')\nprint(f'Train labels loaded: {len(train_labels)} samples')\nprint(train_labels.head())\n\n# Generate image file paths with nested folder structure\ndef get_image_path(image_id):\n    return f'/kaggle/input/bms-molecular-translation/train/{image_id[0]}/{image_id[1]}/{image_id[2]}/{image_id}.png'\n\ntrain_labels['image_path'] = train_labels['image_id'].apply(get_image_path)\n\n# Verify paths exist (check first few)\nprint('\\nChecking if image paths exist (first 5):')\nfor i in range(min(5, len(train_labels))):\n    path = train_labels.iloc[i]['image_path']\n    exists = os.path.exists(path)\n    print(f'{path}: {exists}')\n\nprint(f'\\nDataFrame shape: {train_labels.shape}')\nprint(f'Columns: {list(train_labels.columns)}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T09:10:29.210463Z","iopub.execute_input":"2025-10-14T09:10:29.210689Z","iopub.status.idle":"2025-10-14T09:10:35.466405Z","shell.execute_reply.started":"2025-10-14T09:10:29.210673Z","shell.execute_reply":"2025-10-14T09:10:35.465621Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 3: Build Character-Level Tokenizer for InChI\n# Create character vocabulary from all InChI strings\n\nall_inchi = train_labels['InChI'].tolist()\n\n# Build character set\nchar_set = set()\nfor inchi in all_inchi:\n    char_set.update(inchi)\n\n# Create character to index mapping (add special tokens)\nchar_to_idx = {'<PAD>': 0, '<START>': 1, '<END>': 2, '<UNK>': 3}\nfor idx, char in enumerate(sorted(char_set), start=4):\n    char_to_idx[char] = idx\n\nidx_to_char = {v: k for k, v in char_to_idx.items()}\nvocab_size = len(char_to_idx)\n\nprint(f'Vocabulary size: {vocab_size}')\nprint(f'First 20 characters: {list(char_to_idx.keys())[:20]}')\n\n# Function to encode InChI strings for TEACHER FORCING\ndef encode_inchi_input(inchi, max_length=275):\n    \"\"\"Encode input sequence (starts with <START>, no <END>)\"\"\"\n    encoded = [char_to_idx['<START>']]\n    for char in inchi:\n        encoded.append(char_to_idx.get(char, char_to_idx['<UNK>']))\n    \n    # Pad or truncate\n    if len(encoded) < max_length:\n        encoded.extend([char_to_idx['<PAD>']] * (max_length - len(encoded)))\n    else:\n        encoded = encoded[:max_length]\n    \n    return encoded\n\ndef encode_inchi_target(inchi, max_length=275):\n    \"\"\"Encode target sequence (shifted by one, ends with <END>)\"\"\"\n    encoded = []\n    for char in inchi:\n        encoded.append(char_to_idx.get(char, char_to_idx['<UNK>']))\n    encoded.append(char_to_idx['<END>'])\n    \n    # Pad or truncate\n    if len(encoded) < max_length:\n        encoded.extend([char_to_idx['<PAD>']] * (max_length - len(encoded)))\n    else:\n        encoded = encoded[:max_length]\n    \n    return encoded\n\n# Function to decode sequences back to InChI\ndef decode_inchi(encoded_seq):\n    decoded = []\n    for idx in encoded_seq:\n        if idx == char_to_idx['<END>'] or idx == char_to_idx['<PAD>']:\n            break\n        if idx != char_to_idx['<START>']:\n            decoded.append(idx_to_char.get(idx, '<UNK>'))\n    return ''.join(decoded)\n\n# Test encoding and decoding\ntest_inchi = train_labels.iloc[0]['InChI']\nencoded_input = encode_inchi_input(test_inchi)\nencoded_target = encode_inchi_target(test_inchi)\ndecoded = decode_inchi(encoded_target)\n\nprint(f'\\nOriginal InChI: {test_inchi[:100]}...')\nprint(f'Encoded input length: {len(encoded_input)}')\nprint(f'Encoded target length: {len(encoded_target)}')\nprint(f'Decoded InChI: {decoded[:100]}...')\nprint(f'Match: {test_inchi == decoded}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T09:10:35.470634Z","iopub.execute_input":"2025-10-14T09:10:35.470914Z","iopub.status.idle":"2025-10-14T09:10:38.186352Z","shell.execute_reply.started":"2025-10-14T09:10:35.470892Z","shell.execute_reply":"2025-10-14T09:10:38.185716Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 4: Image Preprocessing and Data Preparation\n# Prepare a smaller subset for faster training (use first 10000 samples)\n# For full training, remove the subset limitation\n\nMAX_SAMPLES = 100  # Reduce for testing, set to None for full training\nIMAGE_SIZE = (224, 224)\nMAX_INCHI_LENGTH = 275\n\nif MAX_SAMPLES:\n    train_labels_subset = train_labels.head(MAX_SAMPLES).copy()\nelse:\n    train_labels_subset = train_labels.copy()\n\nprint(f'Using {len(train_labels_subset)} samples for training')\n\n# Encode all InChI strings\ntrain_labels_subset['encoded_inchi_input'] = train_labels_subset['InChI'].apply(\n    lambda x: encode_inchi_input(x, MAX_INCHI_LENGTH)\n)\ntrain_labels_subset['encoded_inchi_target'] = train_labels_subset['InChI'].apply(\n    lambda x: encode_inchi_target(x, MAX_INCHI_LENGTH)\n)\n\n# Split data: 90% train, 10% validation\ntrain_df, val_df = train_test_split(\n    train_labels_subset, \n    test_size=0.1, \n    random_state=42\n)\n\nprint(f'Train samples: {len(train_df)}')\nprint(f'Validation samples: {len(val_df)}')\n\n# ImageNet normalization constants\nIMAGENET_MEAN = np.array([0.485, 0.456, 0.406])\nIMAGENET_STD = np.array([0.229, 0.224, 0.225])\n\ndef preprocess_image(image_path):\n    \"\"\"Load and preprocess image with ImageNet normalization\"\"\"\n    try:\n        img = cv2.imread(image_path)\n        if img is None:\n            return None\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img = cv2.resize(img, IMAGE_SIZE)\n        img = img.astype(np.float32) / 255.0\n        img = (img - IMAGENET_MEAN) / IMAGENET_STD\n        return img\n    except:\n        return None\n\nprint('\\nData preparation complete!')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T09:10:38.187163Z","iopub.execute_input":"2025-10-14T09:10:38.187444Z","iopub.status.idle":"2025-10-14T09:10:38.201746Z","shell.execute_reply.started":"2025-10-14T09:10:38.187415Z","shell.execute_reply":"2025-10-14T09:10:38.201069Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 5: Create Data Generator with Teacher Forcing\nclass DataGenerator(keras.utils.Sequence):\n    def __init__(self, dataframe, batch_size=32, shuffle=True):\n        self.dataframe = dataframe.reset_index(drop=True)\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.indexes = np.arange(len(self.dataframe))\n        self.on_epoch_end()\n    \n    def __len__(self):\n        return int(np.ceil(len(self.dataframe) / self.batch_size))\n    \n    def __getitem__(self, index):\n        # Get batch indexes\n        batch_indexes = self.indexes[index * self.batch_size:(index + 1) * self.batch_size]\n        \n        # Get batch data\n        images = []\n        decoder_inputs = []\n        targets = []\n        \n        for idx in batch_indexes:\n            row = self.dataframe.iloc[idx]\n            img = preprocess_image(row['image_path'])\n            if img is not None:\n                images.append(img)\n                decoder_inputs.append(row['encoded_inchi_input'])\n                targets.append(row['encoded_inchi_target'])\n        \n        if len(images) == 0:\n            print('error loading the data.')\n            # Return dummy batch if all images failed to load\n            return ({\n                'image_input': np.zeros((1, 224, 224, 3), dtype=np.float32),\n                'decoder_input': np.zeros((1, MAX_INCHI_LENGTH), dtype=np.int32)\n            }, np.zeros((1, MAX_INCHI_LENGTH), dtype=np.int32))\n        \n        return ({\n            'image_input': np.array(images, dtype=np.float32),\n            'decoder_input': np.array(decoder_inputs, dtype=np.int32)\n        }, np.array(targets, dtype=np.int32))\n    \n    def on_epoch_end(self):\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n\nprint('Data generator created successfully')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T09:10:38.202588Z","iopub.execute_input":"2025-10-14T09:10:38.203314Z","iopub.status.idle":"2025-10-14T09:10:38.213613Z","shell.execute_reply.started":"2025-10-14T09:10:38.203289Z","shell.execute_reply":"2025-10-14T09:10:38.213004Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 6: Build PROPER Encoder-Decoder Model with Teacher Forcing\ndef build_model(vocab_size, max_length, learning_rate=1e-4):\n    # IMAGE ENCODER: EfficientNet-B0 pretrained on ImageNet\n    base_model = keras.applications.EfficientNetB0(\n        include_top=False,\n        weights='imagenet',\n        input_shape=(224, 224, 3),\n        pooling='avg'\n    )\n    \n    # Fine-tune the last layers\n    base_model.trainable = True\n    \n    # Image input\n    image_input = layers.Input(shape=(224, 224, 3), name='image_input')\n    \n    # Extract image features\n    image_features = base_model(image_input)\n    image_features = layers.Dense(512, activation='relu', name='image_dense')(image_features)\n    image_features = layers.Dropout(0.3)(image_features)\n    \n    # DECODER INPUT: Previous tokens (for teacher forcing)\n    decoder_input = layers.Input(shape=(max_length,), name='decoder_input')\n    \n    # Embedding layer for decoder input\n    decoder_embedding = layers.Embedding(\n        input_dim=vocab_size,\n        output_dim=256,\n        mask_zero=True,\n        name='decoder_embedding'\n    )(decoder_input)\n    \n    # Initialize decoder state with image features\n    # Repeat image features for each LSTM unit\n    initial_state_h = layers.Dense(512, name='init_state_h')(image_features)\n    initial_state_c = layers.Dense(512, name='init_state_c')(image_features)\n    \n    # LSTM Decoder with initial state from image\n    lstm_out = layers.LSTM(\n        512,\n        return_sequences=True,\n        return_state=False,\n        name='decoder_lstm_1'\n    )(decoder_embedding, initial_state=[initial_state_h, initial_state_c])\n    \n    lstm_out = layers.Dropout(0.3)(lstm_out)\n    \n    # Second LSTM layer\n    lstm_out = layers.LSTM(\n        512,\n        return_sequences=True,\n        name='decoder_lstm_2'\n    )(lstm_out)\n    \n    lstm_out = layers.Dropout(0.3)(lstm_out)\n    \n    # Output layer\n    outputs = layers.Dense(vocab_size, activation='softmax', name='output')(lstm_out)\n\n    # Build model\n    model = keras.Model(\n        inputs=[image_input, decoder_input],\n        outputs=outputs,\n        name='image_to_inchi_encoder_decoder'\n    )\n    \n    # Compile model\n    model.compile(\n        optimizer=keras.optimizers.Adam(learning_rate=learning_rate),\n        loss='sparse_categorical_crossentropy',\n        metrics=['accuracy', LevenshteinDistanceMetric(name='mean_levenshtein_distance')]\n    )\n    \n    return model\n\nprint('Model architecture defined successfully')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T09:10:38.214261Z","iopub.execute_input":"2025-10-14T09:10:38.214434Z","iopub.status.idle":"2025-10-14T09:10:38.232388Z","shell.execute_reply.started":"2025-10-14T09:10:38.214420Z","shell.execute_reply":"2025-10-14T09:10:38.231693Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 7: Levenshtein Distance for Evaluation\nimport Levenshtein\n\nclass LevenshteinDistanceMetric(keras.metrics.Metric):\n    \"\"\"\n    Custom Keras metric to calculate mean Levenshtein distance\n    This will be used in model.compile() for automatic tracking\n    \"\"\"\n    def __init__(self, name='mean_levenshtein_distance', **kwargs):\n        super().__init__(name=name, **kwargs)\n        self.total_distance = self.add_weight(name='total_distance', initializer='zeros')\n        self.count = self.add_weight(name='count', initializer='zeros')\n    \n    def update_state(self, y_true, y_pred, sample_weight=None):\n        \"\"\"\n        Update metric state with batch predictions\n        \n        Note: This is a simplified version that works with token-level accuracy.\n        For exact Levenshtein distance, we need the callback (which does full decoding).\n        This metric provides a proxy that's correlated with Levenshtein distance.\n        \"\"\"\n        # Get predicted tokens (argmax over vocabulary dimension)\n        y_pred_tokens = tf.argmax(y_pred, axis=-1)\n        \n        # Compare with true tokens (element-wise)\n        # This gives us a per-position accuracy, which correlates with Levenshtein\n        matches = tf.cast(tf.equal(y_pred_tokens, tf.cast(y_true, tf.int64)), tf.float32)\n        \n        # Calculate error rate (1 - accuracy) as proxy for edit distance\n        # Higher error rate ≈ higher Levenshtein distance\n        errors_per_sequence = tf.reduce_sum(1.0 - matches, axis=-1)\n        \n        # Update running totals\n        batch_distance = tf.reduce_sum(errors_per_sequence)\n        self.total_distance.assign_add(batch_distance)\n        self.count.assign_add(tf.cast(tf.shape(y_true)[0], tf.float32))\n    \n    def result(self):\n        \"\"\"Return mean distance\"\"\"\n        return tf.math.divide_no_nan(self.total_distance, self.count)\n    \n    def reset_state(self):\n        \"\"\"Reset metric state\"\"\"\n        self.total_distance.assign(0.0)\n        self.count.assign(0.0)\n\nprint('Levenshtein distance metric class defined')\n\n\n# Step 7: Custom Callback for TRUE Levenshtein Distance Validation\nclass MeanLevenshteinCallback(keras.callbacks.Callback):\n    \"\"\"\n    Custom callback to calculate TRUE mean Levenshtein distance on validation set\n    This does full autoregressive decoding and calculates actual edit distance\n    \n    This is more accurate than the compiled metric (which is a proxy)\n    Use this for model selection and early stopping\n    \"\"\"\n    def __init__(self, validation_data, val_df, max_length=275):\n        super().__init__()\n        self.validation_data = validation_data\n        self.val_df = val_df.reset_index(drop=True)\n        self.max_length = max_length\n        self.levenshtein_history = []\n        self.best_distance = float('inf')\n        \n    def on_epoch_end(self, epoch, logs=None):\n        # Sample a subset of validation data for speed (use 10 samples)\n        # For full validation, remove the sampling\n        sample_size = min(10, len(self.val_df))\n        sample_indices = np.random.choice(len(self.val_df), sample_size, replace=False)\n        \n        predictions = []\n        ground_truths = []\n        \n        for idx in sample_indices:\n            row = self.val_df.iloc[idx]\n            img = preprocess_image(row['image_path'])\n            \n            if img is not None:\n                # Generate prediction autoregressively\n                decoder_input = np.zeros((1, self.max_length), dtype=np.int32)\n                decoder_input[0, 0] = char_to_idx['<START>']\n                img_batch = np.expand_dims(img, axis=0)\n                \n                for i in range(1, self.max_length):\n                    preds = self.model.predict([img_batch, decoder_input], verbose=0)\n                    next_token = np.argmax(preds[0, i-1, :])\n                    \n                    if next_token == char_to_idx['<END>'] or next_token == char_to_idx['<PAD>']:\n                        break\n                    \n                    decoder_input[0, i] = next_token\n                \n                pred_str = decode_inchi(decoder_input[0])\n                predictions.append(pred_str)\n                ground_truths.append(row['InChI'])\n        \n        # Calculate TRUE average Levenshtein distance\n        if len(predictions) > 0:\n            distances = [Levenshtein.distance(pred, gt) for pred, gt in zip(predictions, ground_truths)]\n            avg_distance = np.mean(distances)\n            self.levenshtein_history.append(avg_distance)\n            \n            # Update logs with TRUE Levenshtein distance (overrides proxy metric)\n            # Use 'val_mean_levenshtein' to match the validation metric name\n            logs['val_mean_levenshtein'] = avg_distance\n            \n            # Track best distance\n            if avg_distance < self.best_distance:\n                self.best_distance = avg_distance\n            \n            print(f'\\n  TRUE Mean Levenshtein Distance: {avg_distance:.2f} (best: {self.best_distance:.2f})')\n\nprint('Mean Levenshtein callback defined')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T09:10:38.233105Z","iopub.execute_input":"2025-10-14T09:10:38.233294Z","iopub.status.idle":"2025-10-14T09:10:38.266298Z","shell.execute_reply.started":"2025-10-14T09:10:38.233280Z","shell.execute_reply":"2025-10-14T09:10:38.265493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 8: Hyperparameter Grid Search Training\n# Grid search over learning rate and batch size\n\n# Define hyperparameter grid\n# Grid size can be increased when we have more compute.\nparam_grid = {\n    'learning_rate': [1e-3],\n    'batch_size': [32]\n}\n\nbest_score = float('inf')\nbest_params = None\nbest_model = None\n\nprint('Starting hyperparameter grid search...')\nprint(f'Grid: {param_grid}')\nprint(f'\\nTesting {len(param_grid[\"learning_rate\"]) * len(param_grid[\"batch_size\"])} configurations')\n\nfor lr in param_grid['learning_rate']:\n    for bs in param_grid['batch_size']:\n        print(f'\\n=== Training with lr={lr}, batch_size={bs} ===')\n        \n        # Build model\n        model = build_model(vocab_size, MAX_INCHI_LENGTH, learning_rate=lr)\n        \n        # Create data generators\n        train_gen = DataGenerator(train_df, batch_size=bs, shuffle=True)\n        val_gen = DataGenerator(val_df, batch_size=bs, shuffle=False)\n        \n        # Callbacks with TRUE Mean Levenshtein distance monitoring\n        mean_levenshtein_callback = MeanLevenshteinCallback(\n            validation_data=val_gen,\n            val_df=val_df,\n            max_length=MAX_INCHI_LENGTH\n        )\n        \n        early_stopping = keras.callbacks.EarlyStopping(\n            monitor='val_mean_levenshtein',  # Monitor Levenshtein distance instead of loss\n            patience=3,\n            restore_best_weights=True,\n            mode='min'  # Lower distance is better\n        )\n        \n        reduce_lr = keras.callbacks.ReduceLROnPlateau(\n            monitor='val_mean_levenshtein',  # Monitor Levenshtein distance\n            factor=0.5,\n            patience=2,\n            min_lr=1e-6,\n            mode='min'\n        )\n        \n        # Train model - INCREASED EPOCHS\n        history = model.fit(\n            train_gen,\n            validation_data=val_gen,\n            epochs=2,  # Increased from 2\n            callbacks=[mean_levenshtein_callback, early_stopping, reduce_lr],\n            verbose=1\n        )\n        \n        # Evaluate on validation set using Levenshtein distance\n        best_distance = min(history.history['val_mean_levenshtein'])\n        print(f'Best Levenshtein distance: {best_distance:.2f}')\n        \n        # Update best configuration based on Levenshtein distance\n        if best_distance < best_score:\n            best_score = best_distance\n            best_params = {'learning_rate': lr, 'batch_size': bs}\n            best_model = model\n            print(f'New best configuration found!')\n\n\nprint(f'\\n=== Grid Search Complete ===')\nprint(f'Best parameters: {best_params}')\nprint(f'Best validation distance/loss: {best_score:.4f}')\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T09:11:03.362378Z","iopub.execute_input":"2025-10-14T09:11:03.362931Z","iopub.status.idle":"2025-10-14T09:21:51.945347Z","shell.execute_reply.started":"2025-10-14T09:11:03.362909Z","shell.execute_reply":"2025-10-14T09:21:51.944669Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(train_labels_subset)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T09:21:51.946679Z","iopub.execute_input":"2025-10-14T09:21:51.947256Z","iopub.status.idle":"2025-10-14T09:21:51.952403Z","shell.execute_reply.started":"2025-10-14T09:21:51.947234Z","shell.execute_reply":"2025-10-14T09:21:51.951785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 8.5: Final Retraining on Combined Train + Validation Data\nprint('\\n=== Step 8.5: Final Retraining on Combined Data ===')\nprint('Retraining best model on combined train + validation data for maximum performance...')\n\n# Combine train and validation data\nfull_df = train_labels_subset.copy()\nprint(f'Combined dataset size: {len(full_df)} samples')\n\n# Build fresh model with best hyperparameters\nfinal_model = build_model(\n    vocab_size, \n    MAX_INCHI_LENGTH, \n    learning_rate=best_params['learning_rate']\n)\n\n# Create data generator for combined data\nfull_df_gen = DataGenerator(full_df, batch_size=best_params['batch_size'], shuffle=True)\n\n# Train on combined data (no validation split)\n# Use fewer epochs since we already validated the hyperparameters\nprint(f'Training with best hyperparameters: {best_params}')\n\nfinal_history = final_model.fit(\n    full_df_gen,\n    epochs=10,  # Same number of epochs as before\n    verbose=1\n)\n\nprint('\\nFinal retraining complete!')\nprint(f'Final mean_levenshtein: {final_history.history[\"mean_levenshtein_distance\"][-1]:.4f}')\n\n# Use the final model for predictions\nbest_model = final_model\nprint('Updated best_model to final retrained model')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T09:21:51.953190Z","iopub.execute_input":"2025-10-14T09:21:51.953490Z","iopub.status.idle":"2025-10-14T09:23:09.340114Z","shell.execute_reply.started":"2025-10-14T09:21:51.953465Z","shell.execute_reply":"2025-10-14T09:23:09.339325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Step 9: Generate Predictions with AUTOREGRESSIVE DECODING\n# Load test data\ntest_df = pd.read_csv('/kaggle/input/bms-molecular-translation/sample_submission.csv')\nprint(f'Test samples: {len(test_df)}')\n\nif MAX_SAMPLES:\n    test_df = test_df.head(MAX_SAMPLES).copy()\n\n\n# Generate test image paths\ndef get_test_image_path(image_id):\n    return f'/kaggle/input/bms-molecular-translation/test/{image_id[0]}/{image_id[1]}/{image_id[2]}/{image_id}.png'\n\ntest_df['image_path'] = test_df['image_id'].apply(get_test_image_path)\n\n# Verify a few test paths\nprint('\\nVerifying test image paths (first 3):')\nfor i in range(min(3, len(test_df))):\n    path = test_df.iloc[i]['image_path']\n    exists = os.path.exists(path)\n    print(f'{path}: {exists}')\n\n# AUTOREGRESSIVE PREDICTION FUNCTION\ndef predict_inchi_autoregressive(model, image, max_length=275):\n    \"\"\"\n    Generate InChI string autoregressively (one token at a time)\n    [SLOW - Use for single predictions only]\n    \"\"\"\n    # Start with <START> token\n    decoder_input = np.zeros((1, max_length), dtype=np.int32)\n    decoder_input[0, 0] = char_to_idx['<START>']\n    \n    # Expand image dimensions\n    img_batch = np.expand_dims(image, axis=0)\n    \n    # Generate tokens one by one\n    for i in range(1, max_length):\n        # Predict next token\n        predictions = model.predict([img_batch, decoder_input], verbose=0)\n        \n        # Get the token at position i-1 (we're predicting position i)\n        next_token_probs = predictions[0, i-1, :]\n        next_token = np.argmax(next_token_probs)\n        \n        # If we predict <END> or <PAD>, stop\n        if next_token == char_to_idx['<END>'] or next_token == char_to_idx['<PAD>']:\n            break\n        \n        # Add predicted token to decoder input for next iteration\n        decoder_input[0, i] = next_token\n    \n    # Decode the sequence\n    return decode_inchi(decoder_input[0])\n\n\ndef predict_inchi_batch_fast(model, images, max_length=275):\n    \"\"\"\n    OPTIMIZED: Batch prediction for multiple images\n    \n    Speed improvements:\n    - Processes multiple images simultaneously\n    - Reduces model.predict() calls from N*max_length to max_length\n    - 5-10x faster than sequential prediction\n    \n    Args:\n        model: Trained Keras model\n        images: List or array of preprocessed images\n        max_length: Maximum sequence length\n    \n    Returns:\n        List of decoded InChI strings\n    \"\"\"\n    batch_size = len(images)\n    \n    # Initialize decoder inputs for entire batch\n    decoder_inputs = np.zeros((batch_size, max_length), dtype=np.int32)\n    decoder_inputs[:, 0] = char_to_idx['<START>']\n    \n    # Stack images into batch\n    img_batch = np.array(images)\n    \n    # Track which sequences are still generating (not ended)\n    active_seqs = np.ones(batch_size, dtype=bool)\n    \n    # Generate tokens autoregressively for entire batch\n    for i in range(1, max_length):\n        # Early exit if all sequences have ended\n        if not np.any(active_seqs):\n            break\n        \n        # Predict next tokens for ALL images in batch simultaneously\n        predictions = model.predict([img_batch, decoder_inputs], verbose=0)\n        \n        # Get next token for each sequence (argmax over vocabulary)\n        next_tokens = np.argmax(predictions[:, i-1, :], axis=-1)\n        \n        # Update each sequence\n        for j in range(batch_size):\n            if active_seqs[j]:\n                # Check if this sequence should end\n                if (next_tokens[j] == char_to_idx['<END>'] or \n                    next_tokens[j] == char_to_idx['<PAD>']):\n                    active_seqs[j] = False\n                else:\n                    decoder_inputs[j, i] = next_tokens[j]\n    \n    # Decode all sequences\n    decoded_results = []\n    for j in range(batch_size):\n        decoded_results.append(decode_inchi(decoder_inputs[j]))\n    \n    return decoded_results\n\n# Make predictions on test set\nprint('\\nGenerating predictions on test set with BATCHED autoregressive decoding...')\n\n# Step 1: Load all test images first\nprint('Loading test images...')\ntest_images = []\nvalid_indices = []\nfailed_indices = []\n\nfor idx in tqdm(range(len(test_df)), desc=\"Loading images\"):\n    image_path = test_df.iloc[idx]['image_path']\n    img = preprocess_image(image_path)\n    \n    if img is not None:\n        test_images.append(img)\n        valid_indices.append(idx)\n    else:\n        failed_indices.append(idx)\n\nprint(f'Loaded {len(test_images)} images successfully, {len(failed_indices)} failed')\n\n# Step 2: Predict in batches (MUCH faster!)\nPREDICTION_BATCH_SIZE = 64  \nprint(f'\\nPredicting in batches of {PREDICTION_BATCH_SIZE}...')\n\npredictions = []\nnum_batches = int(np.ceil(len(test_images) / PREDICTION_BATCH_SIZE))\n\nfor batch_idx in tqdm(range(num_batches), desc=\"Predicting batches\"):\n    start_idx = batch_idx * PREDICTION_BATCH_SIZE\n    end_idx = min(start_idx + PREDICTION_BATCH_SIZE, len(test_images))\n    \n    batch_images = test_images[start_idx:end_idx]\n    \n    # Predict entire batch at once\n    batch_predictions = predict_inchi_batch_fast(\n        best_model, \n        batch_images, \n        MAX_INCHI_LENGTH\n    )\n    \n    predictions.extend(batch_predictions)\n\n# Debug first few predictions\nprint('\\nFirst 5 predictions:')\nfor i in range(min(5, len(predictions))):\n    pred = predictions[i]\n    print(f'  {i}: {pred[:100]}{\"...\" if len(pred) > 100 else \"\"}')\n    if len(pred) == 0:\n        print(f'    WARNING: Empty prediction!')\n\n# Step 3: Handle failed images and create full prediction list\nfull_predictions = []\nvalid_idx_set = set(valid_indices)\n\nprediction_pointer = 0\nfor idx in range(len(test_df)):\n    if idx in valid_idx_set:\n        pred = predictions[prediction_pointer]\n        # Fallback for empty predictions\n        if len(pred) == 0:\n            pred = 'InChI=1S/C'\n        full_predictions.append(pred)\n        prediction_pointer += 1\n    else:\n        # Use fallback for failed images\n        full_predictions.append('InChI=1S/C')\n\npredictions = full_predictions\n\n# Create submission dataframe\nsubmission = pd.DataFrame({\n    'image_id': test_df['image_id'],\n    'InChI': predictions\n})\n\nprint(f'\\nSubmission shape: {submission.shape}')\nprint(submission.head(10))\n\n# Save submission file\nsubmission.to_csv('submission.csv', index=False)\nprint('\\nSubmission file saved: submission.csv')\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-14T09:23:09.341588Z","iopub.execute_input":"2025-10-14T09:23:09.342085Z","iopub.status.idle":"2025-10-14T09:25:51.817119Z","shell.execute_reply.started":"2025-10-14T09:23:09.342066Z","shell.execute_reply":"2025-10-14T09:25:51.816321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# pred_indices = np.argmax(pred[0], axis=0)\n# decode_inchi(pred_indices)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}