{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.13"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":22962,"databundleVersionId":3171193,"sourceType":"competition"},{"sourceId":3398941,"sourceType":"datasetVersion","datasetId":2040833}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":41.006696,"end_time":"2025-08-03T05:59:48.024918","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-08-03T05:59:07.018222","version":"2.5.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"0242f0ca0559484daa5f53e208b4940d":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"0ad43bed980645c0847df078c41bfa0e":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"","description_tooltip":null,"layout":"IPY_MODEL_8668b357f7ab4259bdd3f60f9d8cea89","max":18,"min":0,"orientation":"horizontal","style":"IPY_MODEL_8933328305d4487da91eca0b59f929d9","value":18}},"0b53c5206eb34f6689cc32900b345a31":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"0d8b8e8413f049c7b1f71e9d63016f2f":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"14a5c260b4e142b79661775e9ab2f2e3":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_0b53c5206eb34f6689cc32900b345a31","placeholder":"​","style":"IPY_MODEL_8f4503703d6c48d095e387d9c761bebc","value":"Epoch 2: 100%"}},"2be52247723d4bfc92b0113b03262625":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":"inline-flex","flex":null,"flex_flow":"row wrap","grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"100%"}},"3f1e58790231494ea54ca1c1dbae1c22":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_e61efd293cd84ff6bc853b34f8e9860c","placeholder":"​","style":"IPY_MODEL_0242f0ca0559484daa5f53e208b4940d","value":"Testing DataLoader 0: 100%"}},"4776a80a5d3c4f1dae8c032fd298aed9":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":"2","flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"4ddfb0468dc84f36b157b401dab0b040":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"7a9d7263f1f1439ba8edb9ea26923257":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_3f1e58790231494ea54ca1c1dbae1c22","IPY_MODEL_0ad43bed980645c0847df078c41bfa0e","IPY_MODEL_f9d0430c993e44598bc43694a8dae3ce"],"layout":"IPY_MODEL_2be52247723d4bfc92b0113b03262625"}},"7cb0b283f32a46529d5d345211178ca4":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":"inline-flex","flex":null,"flex_flow":"row wrap","grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"100%"}},"8668b357f7ab4259bdd3f60f9d8cea89":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":"2","flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8933328305d4487da91eca0b59f929d9":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"8f4503703d6c48d095e387d9c761bebc":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"a9b18ee47ae84012bf67a69167983661":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_f39e84f54d614d64b8c7529b009af830","placeholder":"​","style":"IPY_MODEL_4ddfb0468dc84f36b157b401dab0b040","value":" 70/70 [00:40&lt;00:00,  1.72it/s, v_num=0]"}},"d0d8a7dee29b439084621db10efc20e1":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_14a5c260b4e142b79661775e9ab2f2e3","IPY_MODEL_f693cca8123c4f17830fe2b749632ad6","IPY_MODEL_a9b18ee47ae84012bf67a69167983661"],"layout":"IPY_MODEL_7cb0b283f32a46529d5d345211178ca4"}},"d5eecd7849aa4202b2c5717233b5dd81":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e61efd293cd84ff6bc853b34f8e9860c":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f39e84f54d614d64b8c7529b009af830":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f693cca8123c4f17830fe2b749632ad6":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"","description_tooltip":null,"layout":"IPY_MODEL_4776a80a5d3c4f1dae8c032fd298aed9","max":70,"min":0,"orientation":"horizontal","style":"IPY_MODEL_f6eb65a6c7aa4cf8882db6dc8fa3a295","value":70}},"f6eb65a6c7aa4cf8882db6dc8fa3a295":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"f9d0430c993e44598bc43694a8dae3ce":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_d5eecd7849aa4202b2c5717233b5dd81","placeholder":"​","style":"IPY_MODEL_0d8b8e8413f049c7b1f71e9d63016f2f","value":" 18/18 [00:07&lt;00:00,  2.44it/s]"}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.007357,"end_time":"2025-08-03T05:59:11.600526","exception":false,"start_time":"2025-08-03T05:59:11.593169","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Whale Identification using CNN Feature Extraction**\n**Whale Identification using Pre-trained CNN Feature Extraction and Similarity Search**\n\nhttps://www.kaggle.com/competitions/happy-whale-and-dolphin (2022)","metadata":{"papermill":{"duration":0.005626,"end_time":"2025-08-03T05:59:11.612345","exception":false,"start_time":"2025-08-03T05:59:11.606719","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Introduction\n\nThis notebook demonstrates a transfer learning approach without training for marine mammal identification. Instead of training a new model or fine-tuning existing ones, we leverage a pre-trained EfficientNetB0 model (trained on ImageNet) as a frozen feature extractor, then use similarity search to identify individuals.\n\n","metadata":{}},{"cell_type":"markdown","source":"# Overall Pipeline \n\n## 1. Pre-trained CNN Feature Extraction (No Training)\n- Uses **pre-trained** EfficientNetB0 (trained on ImageNet) as a **frozen feature extractor**\n- **No model training or fine-tuning is performed**\n- Removes the classification head and uses Global Average Pooling\n- Processes images to 224x224 resolution with EfficientNet's preprocessing\n- Extracts 1280-dimensional feature vectors for each image using **inference only**\n\n## 2. Data Loading and Feature Extraction\n- Loads training images from directory structure (organized by individual ID)\n- Processes each image through the **frozen pre-trained model** (inference only)\n- Stores extracted features in a matrix (n_samples × 1280)\n- Encodes string labels into numerical values for evaluation\n- **No gradient computation or weight updates occur**\n\n## 3. Similarity-Based Prediction (No Learning Algorithm)\n- Implements cosine similarity for comparing feature vectors\n- Normalizes vectors before comparison\n- Returns top-k most similar individuals from training set using **nearest neighbor search**\n- Handles cases where no good matches are found\n- **This is pure similarity search, not machine learning training**\n\n## 4. Evaluation Framework\n- Implements top-k accuracy metric (k=5 by default)\n- Tests how often the true individual appears in top predictions\n- Provides quantitative measure of similarity search performance\n- **Evaluates retrieval accuracy, not trained model accuracy**\n\n## 5. Test Prediction and Submission\n- Processes test images through the same **frozen feature extractor**\n- For each test image, finds 5 most similar training images using cosine similarity\n- Generates submission file in required competition format\n- Handles edge cases with placeholder predictions\n\n## Key Points:\n- **No training/learning occurs** - this is purely a **transfer learning feature extraction + similarity search** approach\n- **No epochs, learning rates, or optimization** involved\n- **No backpropagation or weight updates**\n- Uses pre-trained CNN as a **fixed feature encoder**\n- Prediction is based on **nearest neighbor retrieval** in feature space\n- Computationally efficient but limited by the quality of pre-trained features for this specific domain\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import GlobalAveragePooling2D\nfrom sklearn.preprocessing import LabelEncoder\nfrom tensorflow.keras.preprocessing import image\nfrom tensorflow.keras.applications.efficientnet import preprocess_input\nfrom tqdm import tqdm","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Feature Extraction with CNN\ndef create_feature_extractor():\n    \"\"\"Create CNN-based feature extractor using EfficientNet\"\"\"\n    base_model = EfficientNetB0(weights='imagenet', include_top=False)\n    x = GlobalAveragePooling2D()(base_model.output)\n    model = Model(inputs=base_model.input, outputs=x)\n    return model\n\ndef load_and_preprocess(img_path, target_size=(224, 224)):\n    \"\"\"Load and preprocess image for EfficientNet\"\"\"\n    img = image.load_img(img_path, target_size=target_size)\n    x = image.img_to_array(img)\n    x = np.expand_dims(x, axis=0)\n    x = preprocess_input(x)\n    return x\n\ndef extract_features(img_path, model, target_size=(224, 224)):\n    try:\n        # Add explicit color mode for grayscale images\n        img = image.load_img(img_path, target_size=target_size, color_mode='rgb')\n        x = image.img_to_array(img)\n        x = np.expand_dims(x, axis=0)\n        x = preprocess_input(x)\n        \n        # Verify preprocessed image\n        if np.all(x == 0):\n            print(f\"Warning: Zero image after preprocessing - {img_path}\")\n            return None\n            \n        features = model.predict(x, verbose=0).flatten()\n        \n        # Add feature validation\n        if np.all(features == 0):\n            print(f\"Warning: Zero features - {img_path}\")\n            return None\n            \n        return features\n        \n    except Exception as e:\n        print(f\"Error processing {img_path}: {str(e)}\")\n        return None","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. Load and Process Training Data\ndef load_data(csv_path, img_dir, model, sample_limit=None):\n    df = pd.read_csv(csv_path)\n    if sample_limit:\n        df = df.groupby('individual_id').apply(\n            lambda x: x.sample(min(len(x), sample_limit))\n        ).reset_index(drop=True)\n    \n    features = []\n    labels = []\n    filenames = []\n    \n    for _, row in tqdm(df.iterrows(), total=len(df), desc=\"extract features ongoing\"):\n        img_path = os.path.join(img_dir, row['image'])\n        feature = extract_features(img_path, model)\n        \n        if feature is not None:\n            features.append(feature)\n            labels.append(row['individual_id'])\n            filenames.append(row['image'])\n    \n    return np.array(features), np.array(labels), filenames\n\n# Initialize feature extractor\nfeature_model = create_feature_extractor()\n\n# Load training data\ncsv_path = '/kaggle/input/happy-whale-and-dolphin/train.csv'\nimg_dir = '/kaggle/input/happywhale-cropped-removebackground-v1/removedBackground_train_images'\nX_train, y_train, train_files = load_data(csv_path, img_dir, feature_model, sample_limit=None)\n\n# Encode labels\nlabel_encoder = LabelEncoder()\ny_train_encoded = label_encoder.fit_transform(y_train)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3. Similarity Search Function\ndef find_top_k_similar(query_features, candidate_features, candidate_labels, k=5):\n    # Convert all to numpy arrays\n    query_features = np.array(query_features) if not isinstance(query_features, np.ndarray) else query_features\n    candidate_features = np.array(candidate_features) if not isinstance(candidate_features, np.ndarray) else candidate_features\n    \n    # Validate inputs\n    if query_features.size == 0 or candidate_features.size == 0:\n        return ['new_individual'] * k, [0] * k\n    \n    # Reshape features\n    query_features = query_features.reshape(1, -1)\n    candidate_features = candidate_features.reshape(-1, query_features.shape[1])\n    \n    # Alternative similarity metric if cosine fails\n    try:\n        # Cosine similarity\n        query_norm = query_features / np.linalg.norm(query_features, axis=1, keepdims=True)\n        candidates_norm = candidate_features / np.linalg.norm(candidate_features, axis=1, keepdims=True)\n        similarities = np.dot(candidates_norm, query_norm.T).flatten()\n    except:\n        # Fallback to Euclidean distance\n        distances = np.linalg.norm(candidate_features - query_features, axis=1)\n        similarities = 1 / (1 + distances)  # Convert distance to similarity\n        \n    # Get top predictions\n    top_k_indices = np.argsort(similarities)[::-1][:k]\n    return [candidate_labels[i] for i in top_k_indices], [similarities[i] for i in top_k_indices]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 4. Evaluation on Validation Set\ndef evaluate_top_k_accuracy(X_val, y_val, X_train, y_train, k=5):\n    \"\"\"Evaluate top-k accuracy\"\"\"\n    correct = 0\n    total = len(X_val)\n    \n    for i in range(len(X_val)):\n        query_feature = X_val[i]\n        true_label = y_val[i]\n        \n        pred_labels, _ = find_top_k_similar(\n            query_feature, X_train, y_train, k=k\n        )\n        \n        if true_label in pred_labels:\n            correct += 1\n    \n    accuracy = correct / total\n    print(f\"Top-{k} Accuracy: {accuracy:.4f}\")\n    return accuracy","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Test feature extraction on a single image\ntest_img_path = \"/kaggle/input/happywhale-cropped-removebackground-v1/removedBackground_test_image/cd50701ae53ed8.jpg\"\n\n# Check image loading\ntry:\n    img = load_and_preprocess(test_img_path)\n    print(\"Image loaded successfully. Shape:\", img.shape)\nexcept Exception as e:\n    print(\"Image loading failed:\", str(e))\n\n# Check feature extraction\nfeatures = extract_features(test_img_path, feature_model)\nprint(\"Feature shape:\", features.shape if features is not None else \"None\")\nprint(\"Feature sample:\", features[:5] if features is not None else \"None\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 5. Test Prediction and Submission\ndef generate_submission(test_dir, model, X_train, y_train):\n    test_predictions = []\n    \n    for img_file in tqdm(os.listdir(test_dir)):\n        img_path = os.path.join(test_dir, img_file)\n        features = extract_features(img_path, model)\n        \n        if features is None:\n            pred_labels = ['new_individual'] * 5\n        else:\n            pred_labels, _ = find_top_k_similar(features, X_train, y_train, k=5)\n        \n        test_predictions.append({'image': img_file, 'predictions': ' '.join(pred_labels)})\n    \n    return pd.DataFrame(test_predictions)\n\n# Generate submission\ntest_dir = '/kaggle/input/happywhale-cropped-removebackground-v1/removedBackground_test_image'\nsubmission_df = generate_submission(test_dir, feature_model, X_train, y_train)\nsubmission_df.to_csv('submission.csv', index=False)\nprint(submission_df['predictions'].value_counts())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}