{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.13"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":22962,"databundleVersionId":3171193,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":3398941,"sourceType":"datasetVersion","datasetId":2040833},{"sourceId":255702710,"sourceType":"kernelVersion"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":41.006696,"end_time":"2025-08-03T05:59:48.024918","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-08-03T05:59:07.018222","version":"2.5.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"0242f0ca0559484daa5f53e208b4940d":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"0ad43bed980645c0847df078c41bfa0e":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"","description_tooltip":null,"layout":"IPY_MODEL_8668b357f7ab4259bdd3f60f9d8cea89","max":18,"min":0,"orientation":"horizontal","style":"IPY_MODEL_8933328305d4487da91eca0b59f929d9","value":18}},"0b53c5206eb34f6689cc32900b345a31":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"0d8b8e8413f049c7b1f71e9d63016f2f":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"14a5c260b4e142b79661775e9ab2f2e3":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_0b53c5206eb34f6689cc32900b345a31","placeholder":"​","style":"IPY_MODEL_8f4503703d6c48d095e387d9c761bebc","value":"Epoch 2: 100%"}},"2be52247723d4bfc92b0113b03262625":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":"inline-flex","flex":null,"flex_flow":"row wrap","grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"100%"}},"3f1e58790231494ea54ca1c1dbae1c22":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_e61efd293cd84ff6bc853b34f8e9860c","placeholder":"​","style":"IPY_MODEL_0242f0ca0559484daa5f53e208b4940d","value":"Testing DataLoader 0: 100%"}},"4776a80a5d3c4f1dae8c032fd298aed9":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":"2","flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"4ddfb0468dc84f36b157b401dab0b040":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"7a9d7263f1f1439ba8edb9ea26923257":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_3f1e58790231494ea54ca1c1dbae1c22","IPY_MODEL_0ad43bed980645c0847df078c41bfa0e","IPY_MODEL_f9d0430c993e44598bc43694a8dae3ce"],"layout":"IPY_MODEL_2be52247723d4bfc92b0113b03262625"}},"7cb0b283f32a46529d5d345211178ca4":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":"inline-flex","flex":null,"flex_flow":"row wrap","grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"100%"}},"8668b357f7ab4259bdd3f60f9d8cea89":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":"2","flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8933328305d4487da91eca0b59f929d9":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"8f4503703d6c48d095e387d9c761bebc":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"a9b18ee47ae84012bf67a69167983661":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_f39e84f54d614d64b8c7529b009af830","placeholder":"​","style":"IPY_MODEL_4ddfb0468dc84f36b157b401dab0b040","value":" 70/70 [00:40&lt;00:00,  1.72it/s, v_num=0]"}},"d0d8a7dee29b439084621db10efc20e1":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_14a5c260b4e142b79661775e9ab2f2e3","IPY_MODEL_f693cca8123c4f17830fe2b749632ad6","IPY_MODEL_a9b18ee47ae84012bf67a69167983661"],"layout":"IPY_MODEL_7cb0b283f32a46529d5d345211178ca4"}},"d5eecd7849aa4202b2c5717233b5dd81":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e61efd293cd84ff6bc853b34f8e9860c":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f39e84f54d614d64b8c7529b009af830":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f693cca8123c4f17830fe2b749632ad6":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"","description_tooltip":null,"layout":"IPY_MODEL_4776a80a5d3c4f1dae8c032fd298aed9","max":70,"min":0,"orientation":"horizontal","style":"IPY_MODEL_f6eb65a6c7aa4cf8882db6dc8fa3a295","value":70}},"f6eb65a6c7aa4cf8882db6dc8fa3a295":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"f9d0430c993e44598bc43694a8dae3ce":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_d5eecd7849aa4202b2c5717233b5dd81","placeholder":"​","style":"IPY_MODEL_0d8b8e8413f049c7b1f71e9d63016f2f","value":" 18/18 [00:07&lt;00:00,  2.44it/s]"}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.007357,"end_time":"2025-08-03T05:59:11.600526","exception":false,"start_time":"2025-08-03T05:59:11.593169","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Whale Identification with CNN Inference**\n\nhttps://www.kaggle.com/competitions/happy-whale-and-dolphin (2022)\n\nhttps://www.kaggle.com/code/stpeteishii/whale-identification-with-cnn-training\n\nhttps://www.kaggle.com/code/stpeteishii/whale-identification-with-cnn-inference","metadata":{"papermill":{"duration":0.005626,"end_time":"2025-08-03T05:59:11.612345","exception":false,"start_time":"2025-08-03T05:59:11.606719","status":"completed"},"tags":[]},"attachments":{}},{"cell_type":"markdown","source":"---\n\n## **Introduction**\nThis pipeline provides a complete deep learning solution for identifying individual whales from images, designed for the Happywhale competition. The system leverages a fine-tuned **EfficientNetB0** model with special handling for mixed-precision training artifacts and comprehensive inference capabilities.\n\n### **Key Features**\n- **Robust model loading** with multiple fallback strategies\n- **Production-ready inference** for both single images and batch processing\n- **Intelligent prediction handling** with confidence thresholding\n- **Error-resilient design** that gracefully handles problematic images\n- **Submission-ready output** formatted for competition requirements\n\n---\n\n## **Pipeline Stages**\n\n### **1. Model Initialization**\nThe system first loads the trained model through multiple strategies:\n- **Primary load**: Attempts standard Keras loading with custom mixed-precision objects\n- **Fallback #1**: Converts to TensorFlow SavedModel format if direct loading fails\n- **Fallback #2**: Reconstructs model architecture from scratch if needed\n\nConcurrently loads a label encoder for converting numeric predictions to whale IDs.\n\n### **2. Image Processing**\nEach image undergoes careful preprocessing:\n1. **Loading & resizing** to EfficientNet's 224×224 input dimensions\n2. **Normalization** using EfficientNet's specific preprocessing:\n   - Pixel values scaled to [-1, 1] range\n   - Channel-wise mean subtraction\n3. **Batch preparation** by adding a dimension for single-image prediction\n\n### **3. Prediction Generation**\nThe core classification process:\n- **Model inference** produces class probabilities\n- **Top-k selection** extracts the most likely predictions\n- **Confidence thresholding** replaces low-confidence predictions with 'new_individual'\n- **Label conversion** translates numeric outputs to whale IDs when possible\n\n### **4. Batch Processing**\nFor competition submissions:\n- **Automatic directory scanning** finds all test images\n- **Parallel-safe processing** handles each image independently\n- **Error recovery** logs but continues past problematic images\n- **Submission formatting** creates competition-ready CSV with:\n  - One row per image\n  - Space-separated top predictions\n  - Automatic fallback to 'new_individual' where needed\n\n## **Technical Advantages**\n1. **Mixed-Precision Compatibility**\n   - Special handling for models trained with float16/float32 mixed precision\n   - Custom layer definitions to maintain precision during inference\n\n2. **Production Resilience**\n   - Multiple loading fallbacks prevent deployment failures\n   - Per-image error isolation prevents batch crashes\n   - Automatic recovery mechanisms for common issues\n\n3. **Competition Optimization**\n   - Configurable confidence thresholds balance precision/recall\n   - Top-k predictions meet competition submission requirements\n   - Sample analysis helps debug model performance\n\nThis pipeline demonstrates how to transition a research model into a robust production system capable of handling real-world data variability while maintaining competition requirements.\n\n---","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport pickle\nfrom tensorflow.keras.models import load_model\nfrom tensorflow.keras.preprocessing.image import load_img, img_to_array\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\nfrom sklearn.preprocessing import LabelEncoder\nfrom tqdm import tqdm\nimport tensorflow as tf\nfrom tensorflow.keras import mixed_precision\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Enable mixed precision (same as training)\npolicy = mixed_precision.Policy('mixed_float16')\nmixed_precision.set_global_policy(policy)\n\n# Custom objects for model loading (fixes Cast layer issue)\ndef get_custom_objects():\n    \"\"\"Get custom objects needed for loading mixed precision models\"\"\"\n    import tensorflow.keras.utils as keras_utils\n    \n    # Create a Cast layer class that can be serialized\n    @keras_utils.register_keras_serializable(package='Custom')\n    class Cast(tf.keras.layers.Layer):\n        def __init__(self, dtype, **kwargs):\n            super().__init__(**kwargs)\n            self.dtype_name = dtype\n            \n        def call(self, inputs):\n            return tf.cast(inputs, dtype=self.dtype_name)\n            \n        def get_config(self):\n            config = super().get_config()\n            config.update({'dtype': self.dtype_name})\n            return config\n    \n    return {'Cast': Cast}\n\nclass WhaleInference:\n    def __init__(self, model_path, label_encoder_path=None):\n        \"\"\"\n        Initialize the inference pipeline\n        \n        Args:\n            model_path: Path to the trained .h5 model\n            label_encoder_path: Path to saved label encoder (optional)\n        \"\"\"\n        self.model_path = model_path\n        self.label_encoder_path = label_encoder_path\n        self.model = None\n        self.label_encoder = None\n        self.img_size = (224, 224)\n        \n    def load_model_and_encoder(self):\n        \"\"\"Load the trained model and label encoder\"\"\"\n        print(\"Loading model...\")\n        \n        # Try multiple loading approaches to handle mixed precision issues\n        try:\n            # First attempt: Use keras custom object scope\n            with tf.keras.utils.custom_object_scope(get_custom_objects()):\n                self.model = load_model(self.model_path)\n            print(f\"Model loaded successfully with custom object scope from {self.model_path}\")\n        except Exception as e1:\n            print(f\"Custom object scope loading failed: {e1}\")\n            try:\n                # Second attempt: Load with TensorFlow SavedModel format conversion\n                print(\"Trying to load as SavedModel format...\")\n                import tempfile\n                import shutil\n                \n                # Create temporary directory for SavedModel conversion\n                with tempfile.TemporaryDirectory() as temp_dir:\n                    # Try to convert and reload\n                    temp_model_path = os.path.join(temp_dir, 'temp_model')\n                    \n                    # Load with compile=False and save as SavedModel\n                    with tf.keras.utils.custom_object_scope(get_custom_objects()):\n                        temp_model = load_model(self.model_path, compile=False)\n                        temp_model.save(temp_model_path, save_format='tf')\n                    \n                    # Load the SavedModel version\n                    self.model = tf.keras.models.load_model(temp_model_path)\n                    \n                print(f\"Model loaded successfully via SavedModel conversion from {self.model_path}\")\n                \n                # Recompile the model\n                from tensorflow.keras.optimizers import Adam\n                self.model.compile(\n                    optimizer=Adam(learning_rate=1e-4),\n                    loss='categorical_crossentropy',\n                    metrics=['accuracy']\n                )\n                print(\"Model recompiled successfully\")\n                \n            except Exception as e2:\n                print(f\"SavedModel conversion failed: {e2}\")\n                try:\n                    # Third attempt: Manual reconstruction (requires recreating architecture)\n                    print(\"Attempting to load weights only and reconstruct model...\")\n                    self._reconstruct_model_from_weights()\n                    \n                except Exception as e3:\n                    print(f\"All loading attempts failed!\")\n                    print(f\"Error 1 (Custom scope): {e1}\")\n                    print(f\"Error 2 (SavedModel): {e2}\")\n                    print(f\"Error 3 (Reconstruction): {e3}\")\n                    print(\"\\nSuggested solutions:\")\n                    print(\"1. Retrain and save the model without mixed precision\")\n                    print(\"2. Use model.save_weights() in training and reconstruct architecture\")\n                    print(\"3. Save model as SavedModel format during training\")\n                    raise Exception(\"Could not load model. Please check the model file and try recreating it.\")\n        \n        # Load label encoder if available\n        if self.label_encoder_path and os.path.exists(self.label_encoder_path):\n            print(\"Loading label encoder...\")\n            with open(self.label_encoder_path, 'rb') as f:\n                self.label_encoder = pickle.load(f)\n            print(f\"Label encoder loaded with {len(self.label_encoder.classes_)} classes\")\n        else:\n            print(\"Warning: No label encoder found. You'll need to provide class mapping manually.\")\n    \n    def _reconstruct_model_from_weights(self):\n        \"\"\"Reconstruct model architecture and load weights (fallback method)\"\"\"\n        print(\"Reconstructing model architecture...\")\n        \n        # This requires knowing the model architecture\n        # We'll use the same architecture as in training\n        from tensorflow.keras.applications import EfficientNetB0\n        from tensorflow.keras.models import Model\n        from tensorflow.keras.layers import GlobalAveragePooling2D, Dense, Dropout\n        \n        # Try to infer number of classes from weight file\n        # This is a fallback - ideally you'd save this info\n        try:\n            # Create a dummy model to check output shape\n            temp_model = load_model(self.model_path, compile=False)\n            num_classes = temp_model.layers[-1].output_shape[-1]\n            del temp_model\n        except:\n            # Default fallback - you might need to adjust this\n            num_classes = 1000  # Adjust based on your training data\n            print(f\"Using default num_classes={num_classes}. You may need to adjust this.\")\n        \n        # Recreate model architecture (same as training)\n        base_model = EfficientNetB0(\n            weights='imagenet',\n            include_top=False,\n            input_shape=(224, 224, 3),\n            pooling=None\n        )\n        \n        # Custom head (same as training)\n        x = base_model.output\n        x = GlobalAveragePooling2D()(x)\n        x = Dropout(0.3)(x)\n        predictions = Dense(num_classes, activation='softmax', dtype='float32')(x)\n        \n        self.model = Model(inputs=base_model.input, outputs=predictions)\n        \n        # Load weights\n        self.model.load_weights(self.model_path)\n        \n        # Compile model\n        from tensorflow.keras.optimizers import Adam\n        self.model.compile(\n            optimizer=Adam(learning_rate=1e-4),\n            loss='categorical_crossentropy',\n            metrics=['accuracy']\n        )\n        \n        print(\"Model reconstructed and weights loaded successfully\")\n    \n    def preprocess_image(self, img_path):\n        \"\"\"\n        Preprocess a single image for inference\n        \n        Args:\n            img_path: Path to the image\n            \n        Returns:\n            Preprocessed image array\n        \"\"\"\n        try:\n            # Load and resize image\n            img = load_img(img_path, target_size=self.img_size)\n            img_array = img_to_array(img)\n            \n            # Apply EfficientNet preprocessing\n            img_array = preprocess_input(img_array)\n            \n            # Add batch dimension\n            img_array = np.expand_dims(img_array, axis=0)\n            \n            return img_array\n        except Exception as e:\n            print(f\"Error preprocessing image {img_path}: {e}\")\n            return None\n    \n    def predict_single_image(self, img_path, top_k=5):\n        \"\"\"\n        Make prediction for a single image\n        \n        Args:\n            img_path: Path to the image\n            top_k: Number of top predictions to return\n            \n        Returns:\n            List of (class_id, probability) tuples\n        \"\"\"\n        # Preprocess image\n        img_array = self.preprocess_image(img_path)\n        if img_array is None:\n            return []\n        \n        # Make prediction\n        predictions = self.model.predict(img_array, verbose=0)[0]\n        \n        # Get top-k predictions\n        top_indices = np.argsort(predictions)[::-1][:top_k]\n        top_probs = predictions[top_indices]\n        \n        # Convert indices to class names if label encoder available\n        if self.label_encoder is not None:\n            top_classes = self.label_encoder.inverse_transform(top_indices)\n        else:\n            top_classes = top_indices\n        \n        return list(zip(top_classes, top_probs))\n    \n    def predict_test_set(self, test_dir, output_csv='submission.csv', \n                        confidence_threshold=0.1, top_k=5):\n        \"\"\"\n        Make predictions for all images in test directory\n        \n        Args:\n            test_dir: Directory containing test images\n            output_csv: Output CSV file path\n            confidence_threshold: Minimum confidence for valid prediction\n            top_k: Number of predictions per image\n        \"\"\"\n        print(f\"Starting inference on test images in {test_dir}\")\n        \n        # Get all image files\n        image_files = [f for f in os.listdir(test_dir) \n                      if f.lower().endswith(('.jpg', '.jpeg', '.png'))]\n        \n        if not image_files:\n            print(\"No image files found in test directory!\")\n            return\n        \n        print(f\"Found {len(image_files)} test images\")\n        \n        results = []\n        failed_images = []\n        \n        # Process each image\n        for img_file in tqdm(image_files, desc=\"Processing images\"):\n            img_path = os.path.join(test_dir, img_file)\n            \n            try:\n                # Get top-k predictions\n                predictions = self.predict_single_image(img_path, top_k=top_k)\n                \n                if not predictions:\n                    failed_images.append(img_file)\n                    continue\n                \n                # Format predictions for submission\n                pred_list = []\n                for class_id, prob in predictions:\n                    if prob >= confidence_threshold:\n                        pred_list.append(str(class_id))\n                    else:\n                        pred_list.append('new_individual')\n                \n                # Ensure we have exactly top_k predictions\n                while len(pred_list) < top_k:\n                    pred_list.append('new_individual')\n                \n                # Create submission row\n                results.append({\n                    'image': img_file,\n                    'predictions': ' '.join(pred_list[:top_k])\n                })\n                \n            except Exception as e:\n                print(f\"Error processing {img_file}: {e}\")\n                failed_images.append(img_file)\n        \n        # Create submission DataFrame\n        submission_df = pd.DataFrame(results)\n        \n        # Add failed images with default predictions\n        for img_file in failed_images:\n            submission_df = pd.concat([\n                submission_df,\n                pd.DataFrame({\n                    'image': [img_file],\n                    'predictions': [' '.join(['new_individual'] * top_k)]\n                })\n            ], ignore_index=True)\n        \n        # Save submission file\n        submission_df.to_csv(output_csv, index=False)\n        print(f\"\\nInference completed!\")\n        print(f\"Results saved to {output_csv}\")\n        print(f\"Successfully processed: {len(results)} images\")\n        print(f\"Failed to process: {len(failed_images)} images\")\n        \n        return submission_df\n    \n    def analyze_predictions(self, test_dir, sample_size=10):\n        \"\"\"\n        Analyze a sample of predictions for debugging\n        \n        Args:\n            test_dir: Directory containing test images\n            sample_size: Number of sample images to analyze\n        \"\"\"\n        print(\"Analyzing sample predictions...\")\n        \n        image_files = [f for f in os.listdir(test_dir) \n                      if f.lower().endswith(('.jpg', '.jpeg', '.png'))]\n        \n        # Sample random images\n        sample_files = np.random.choice(image_files, \n                                       min(sample_size, len(image_files)), \n                                       replace=False)\n        \n        for img_file in sample_files:\n            img_path = os.path.join(test_dir, img_file)\n            predictions = self.predict_single_image(img_path, top_k=5)\n            \n            print(f\"\\nImage: {img_file}\")\n            print(\"Top 5 predictions:\")\n            for i, (class_id, prob) in enumerate(predictions, 1):\n                print(f\"  {i}. {class_id}: {prob:.4f}\")\n\ndef save_label_encoder(label_encoder, filepath):\n    \"\"\"Save label encoder for later use\"\"\"\n    with open(filepath, 'wb') as f:\n        pickle.dump(label_encoder, f)\n    print(f\"Label encoder saved to {filepath}\")\n\ndef run_inference():\n    \"\"\"\n    Main inference pipeline\n    Modify paths according to your setup\n    \"\"\"\n    # Configuration\n    MODEL_PATH = '/kaggle/input/whale-identification-with-cnn-training/best_model.h5'  # Use the fixed model from converter\n    TEST_DIR = '/kaggle/input/happywhale-cropped-removebackground-v1/removedBackground_test_image'      # Directory containing test images\n    LABEL_ENCODER_PATH = 'label_encoder.pkl'  # Optional: saved label encoder\n    OUTPUT_CSV = 'submission.csv'\n    \n    # Check if model exists\n    if not os.path.exists(MODEL_PATH):\n        print(f\"Error: Model file {MODEL_PATH} not found!\")\n        print(\"Please run the model converter script first, or ensure you have the trained model\")\n        return\n    \n    # Check if test directory exists\n    if not os.path.exists(TEST_DIR):\n        print(f\"Error: Test directory {TEST_DIR} not found!\")\n        print(\"Please create the test_images directory and add your test images\")\n        return\n    \n    # Initialize inference pipeline\n    inference = WhaleInference(\n        model_path=MODEL_PATH,\n        label_encoder_path=LABEL_ENCODER_PATH\n    )\n    \n    # Load model and encoder\n    inference.load_model_and_encoder()\n    \n    # Analyze a few sample predictions (optional)\n    print(\"\\nAnalyzing sample predictions...\")\n    inference.analyze_predictions(TEST_DIR, sample_size=3)\n    \n    # Generate predictions for all test images\n    print(\"\\nGenerating submission file...\")\n    submission_df = inference.predict_test_set(\n        test_dir=TEST_DIR,\n        output_csv=OUTPUT_CSV,\n        confidence_threshold=0.05,  # Lower threshold to avoid too many 'new_individual'\n        top_k=5\n    )\n    \n    # Display sample of results\n    print(\"\\nSample submission format:\")\n    print(submission_df.head(10))\n    \n    print(f\"\\nSubmission file saved as: {OUTPUT_CSV}\")\n    print(\"Format matches competition requirements: image,predictions\")\n\n# Additional utility function to create label encoder from training data\ndef create_label_encoder_from_training_data(csv_path, save_path='label_encoder.pkl'):\n    \"\"\"\n    Create and save label encoder from training data\n    Use this if you don't have the label encoder saved from training\n    \"\"\"\n    print(\"Creating label encoder from training data...\")\n    df = pd.read_csv(csv_path)\n    \n    # Apply same filtering as in training (optional)\n    class_counts = df['individual_id'].value_counts()\n    valid_classes = class_counts[class_counts >= 5].index  # Min 5 samples\n    df = df[df['individual_id'].isin(valid_classes)]\n    \n    # Create and fit label encoder\n    label_encoder = LabelEncoder()\n    label_encoder.fit(df['individual_id'].unique())\n    \n    # Save for inference\n    save_label_encoder(label_encoder, save_path)\n    \n    print(f\"Label encoder created with {len(label_encoder.classes_)} classes\")\n    return label_encoder\n\nif __name__ == \"__main__\":\n    # First, run the model converter if you have the original problematic model\n    # Uncomment these lines if you need to create label encoder from training data:\n    # TRAIN_CSV = '/kaggle/input/happy-whale-and-dolphin/train.csv'\n    # create_label_encoder_from_training_data(TRAIN_CSV)\n    \n    # Run inference\n    run_inference()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}