{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.11.13"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":22962,"databundleVersionId":3171193,"sourceType":"competition"},{"sourceId":3398941,"sourceType":"datasetVersion","datasetId":2040833}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":41.006696,"end_time":"2025-08-03T05:59:48.024918","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2025-08-03T05:59:07.018222","version":"2.5.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"0242f0ca0559484daa5f53e208b4940d":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"0ad43bed980645c0847df078c41bfa0e":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"","description_tooltip":null,"layout":"IPY_MODEL_8668b357f7ab4259bdd3f60f9d8cea89","max":18,"min":0,"orientation":"horizontal","style":"IPY_MODEL_8933328305d4487da91eca0b59f929d9","value":18}},"0b53c5206eb34f6689cc32900b345a31":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"0d8b8e8413f049c7b1f71e9d63016f2f":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"14a5c260b4e142b79661775e9ab2f2e3":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_0b53c5206eb34f6689cc32900b345a31","placeholder":"​","style":"IPY_MODEL_8f4503703d6c48d095e387d9c761bebc","value":"Epoch 2: 100%"}},"2be52247723d4bfc92b0113b03262625":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":"inline-flex","flex":null,"flex_flow":"row wrap","grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"100%"}},"3f1e58790231494ea54ca1c1dbae1c22":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_e61efd293cd84ff6bc853b34f8e9860c","placeholder":"​","style":"IPY_MODEL_0242f0ca0559484daa5f53e208b4940d","value":"Testing DataLoader 0: 100%"}},"4776a80a5d3c4f1dae8c032fd298aed9":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":"2","flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"4ddfb0468dc84f36b157b401dab0b040":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"7a9d7263f1f1439ba8edb9ea26923257":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_3f1e58790231494ea54ca1c1dbae1c22","IPY_MODEL_0ad43bed980645c0847df078c41bfa0e","IPY_MODEL_f9d0430c993e44598bc43694a8dae3ce"],"layout":"IPY_MODEL_2be52247723d4bfc92b0113b03262625"}},"7cb0b283f32a46529d5d345211178ca4":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":"inline-flex","flex":null,"flex_flow":"row wrap","grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":"100%"}},"8668b357f7ab4259bdd3f60f9d8cea89":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":"2","flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"8933328305d4487da91eca0b59f929d9":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"8f4503703d6c48d095e387d9c761bebc":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"a9b18ee47ae84012bf67a69167983661":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_f39e84f54d614d64b8c7529b009af830","placeholder":"​","style":"IPY_MODEL_4ddfb0468dc84f36b157b401dab0b040","value":" 70/70 [00:40&lt;00:00,  1.72it/s, v_num=0]"}},"d0d8a7dee29b439084621db10efc20e1":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_14a5c260b4e142b79661775e9ab2f2e3","IPY_MODEL_f693cca8123c4f17830fe2b749632ad6","IPY_MODEL_a9b18ee47ae84012bf67a69167983661"],"layout":"IPY_MODEL_7cb0b283f32a46529d5d345211178ca4"}},"d5eecd7849aa4202b2c5717233b5dd81":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"e61efd293cd84ff6bc853b34f8e9860c":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f39e84f54d614d64b8c7529b009af830":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"f693cca8123c4f17830fe2b749632ad6":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"","description_tooltip":null,"layout":"IPY_MODEL_4776a80a5d3c4f1dae8c032fd298aed9","max":70,"min":0,"orientation":"horizontal","style":"IPY_MODEL_f6eb65a6c7aa4cf8882db6dc8fa3a295","value":70}},"f6eb65a6c7aa4cf8882db6dc8fa3a295":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"f9d0430c993e44598bc43694a8dae3ce":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_d5eecd7849aa4202b2c5717233b5dd81","placeholder":"​","style":"IPY_MODEL_0d8b8e8413f049c7b1f71e9d63016f2f","value":" 18/18 [00:07&lt;00:00,  2.44it/s]"}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.007357,"end_time":"2025-08-03T05:59:11.600526","exception":false,"start_time":"2025-08-03T05:59:11.593169","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Happy Whale Identification XGBRanker**\nhttps://www.kaggle.com/competitions/happy-whale-and-dolphin (2022)","metadata":{"papermill":{"duration":0.005626,"end_time":"2025-08-03T05:59:11.612345","exception":false,"start_time":"2025-08-03T05:59:11.606719","status":"completed"},"tags":[]}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n---\n\n## 🐋 Whale Identification Using HOG Features and XGBRanker\n\nIdentifying individual whales from photographs is a challenging yet vital task in marine research and conservation. Each whale may be photographed under different lighting, poses, and backgrounds, making traditional classification methods difficult to scale.\n\nIn this notebook, we explore a lightweight and interpretable approach for whale identification using:\n\n* **Histogram of Oriented Gradients (HOG)**: a classical feature extraction method that captures local edge patterns and shape descriptors from grayscale images.\n* **XGBoost Ranker**: a powerful gradient boosting model adapted for **learning to rank**, which allows us to frame the problem as a ranking task rather than a pure classification one.\n\nInstead of training a model to classify each whale directly, we treat identification as a **ranking problem**: given a query image, the model must rank a gallery of known individuals such that the correct match appears among the top results.\n\nThis approach enables us to:\n\n* Work with limited training data and computational resources,\n* Improve robustness to label imbalance,\n* Evaluate performance using **top-k accuracy**, reflecting real-world identification use cases.\n\nThe following sections detail each step of the pipeline, from feature extraction and training data preparation to model training, evaluation, and final prediction for submission.\n\n---\n","metadata":{}},{"cell_type":"code","source":"","metadata":{"papermill":{"duration":0.005583,"end_time":"2025-08-03T05:59:11.624333","exception":false,"start_time":"2025-08-03T05:59:11.61875","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport pandas as pd\nfrom skimage.feature import hog\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.metrics import accuracy_score\nimport xgboost as xgb\nimport os\nimport matplotlib.pyplot as plt\nfrom collections import defaultdict\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and prepare data\nprint(\"Loading training data...\")\ntrain0 = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')\nprint(f\"Original training data columns: {train0.columns.tolist()}\")\nprint(f\"Original training data length: {len(train0)}\")\n\n# Filter individuals with at least 100 samples for better training\nid_counts = train0['individual_id'].value_counts()\nvalid_ids = id_counts[id_counts >= 10].index\ntrain = train0[train0['individual_id'].isin(valid_ids)]\nprint(f\"Filtered training data length: {len(train)}\")\n\n# Create mappings\nfile2id = train.set_index(\"image\")[\"individual_id\"].to_dict()\nunique_ids = sorted(train['individual_id'].unique().tolist())\nprint(f\"Number of unique individuals: {len(unique_ids)}\")\nprint(f\"First 5 individual IDs: {unique_ids[:5]}\")\n\n# Label encoder for individuals\nlabel_encoder = LabelEncoder()\nlabel_encoder.fit(unique_ids)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n---\n\n### **1. `extract_hog_features(image_path)`**\n\n* This function **reads an image**, **resizes it to a standard size**, and **extracts HOG (Histogram of Oriented Gradients) features**.\n* HOG features describe the **direction and intensity of edges** in local regions of the image.\n* These features are useful for tasks like **object or animal identification** because they capture **shape and texture information**.\n\n---\n\n### **2. `load_features_from_directory(directory_path, file_to_id_map, valid_ids)`**\n\n* This function **iterates through all image files in a directory**.\n* It uses a mapping (`file_to_id_map`) to get the **ID of the individual (e.g., whale) in each image**.\n* It **filters images** to include only those with IDs in the given `valid_ids` set.\n* For each valid image, it **extracts HOG features**, then **stores the features, label (ID), and filename**.\n* Finally, it returns arrays of features, labels, and image names — ready for model training or analysis.\n\n---\n\n\n","metadata":{}},{"cell_type":"code","source":"def extract_hog_features(image_path):\n    \"\"\"\n    Extract HOG (Histogram of Oriented Gradients) features from an image.\n    \n    HOG captures local object appearance and shape by analyzing edge directions\n    in localized regions, making it effective for whale identification.\n    \"\"\"\n    try:\n        image = cv2.imread(image_path, cv2.IMREAD_GRAYSCALE)\n        if image is None:\n            print(f\"Warning: Could not read image {image_path}\")\n            return None\n        \n        # Standardize image size\n        image = cv2.resize(image, (128,128))\n        \n        # Extract HOG features\n        features = hog(\n            image,\n            orientations=9,           # Number of orientation bins\n            pixels_per_cell=(8, 8),   # Size of cells\n            cells_per_block=(2, 2),   # Number of cells per block\n            block_norm='L2-Hys',      # Block normalization method\n            transform_sqrt=True,      # Apply power law compression\n            visualize=False\n        )\n        return features\n    except Exception as e:\n        print(f\"Error processing image {image_path}: {str(e)}\")\n        return None\n\n\ndef load_features_from_directory(directory_path, file_to_id_map, valid_ids):\n    \"\"\"Load HOG features from all images in a directory.\"\"\"\n    features = []\n    labels = []\n    image_names = []\n    \n    print(f\"Loading features from: {directory_path}\")\n    \n    for filename in os.listdir(directory_path):\n        if filename.lower().endswith(('.jpg', '.jpeg', '.png')):\n            image_path = os.path.join(directory_path, filename)\n            \n            # Get individual ID for this image\n            if filename in file_to_id_map:\n                individual_id = file_to_id_map[filename]\n                \n                # Only process if this individual is in our valid set\n                if individual_id in valid_ids:\n                    hog_features = extract_hog_features(image_path)\n                    \n                    if hog_features is not None:\n                        features.append(hog_features)\n                        labels.append(individual_id)\n                        image_names.append(filename)\n    \n    return np.array(features), np.array(labels), image_names","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n---\n\n### 🔍 **Overview:**\n\nThis section prepares HOG-based features from training images for use in an XGBoost classifier.\n\n---\n\n### 🐋 **1. Feature Extraction:**\n\n* It loads whale images from a specified training directory.\n* For each valid image, it extracts HOG features using the earlier defined process.\n* It also collects the corresponding individual ID and image filename.\n\n---\n\n### 📊 **2. Label Encoding:**\n\n* Since machine learning models require numeric labels, it converts the whale IDs (strings) into integers using a label encoder.\n\n---\n\n### ✂️ **3. Train/Validation Split:**\n\n* The extracted features and labels are split into a **training set (80%)** and a **validation set (20%)**.\n* The split is **stratified**, meaning it maintains the same label distribution in both sets.\n\n---\n\n### ✅ **Final Output:**\n\n* You now have:\n\n  * HOG features as input (X)\n  * Encoded labels as output (y)\n  * Separate training and validation sets, ready for model training and evaluation.\n\n---\n\n","metadata":{}},{"cell_type":"code","source":"# Load training features\nprint(\"Extracting HOG features from training images...\")\ntrain_dir = '/kaggle/input/happywhale-cropped-removebackground-v1/removedBackground_train_images'\nX_features, y_labels, train_image_names = load_features_from_directory(\n    train_dir, file2id, unique_ids\n)\n\nprint(f\"Loaded {len(X_features)} training samples\")\nprint(f\"Feature dimension: {X_features.shape[1] if len(X_features) > 0 else 'N/A'}\")\n\n# Encode labels for XGBoost\ny_encoded = label_encoder.transform(y_labels)\n\n# Split data for training and validation\nX_train, X_val, y_train, y_val = train_test_split(\n    X_features, y_encoded, test_size=0.2, random_state=42, stratify=y_encoded\n)\n\nprint(f\"Training set: {X_train.shape[0]} samples\")\nprint(f\"Validation set: {X_val.shape[0]} samples\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n---\n\n### 🎯 **Purpose:**\n\nThis block creates **ranking-style training data** for use with **XGBoost Ranker**, which is suited for **identifying the correct match (whale) from a set of candidates**.\n\n---\n\n### 🔁 **What It Does:**\n\n#### 1. **Ranking Problem Construction**:\n\n* For each image in the dataset (up to 1000 samples for efficiency):\n\n  * It treats that image as a **query**.\n  * The correct individual (same label) is marked as **relevant (label = 1)**.\n  * It then randomly selects other individuals (different labels) to serve as **non-relevant candidates (label = 0)**.\n\n#### 2. **Group Formation**:\n\n* Each such set — one positive sample and several negatives — forms a **\"group\"**.\n* These groups are needed by XGBoost Ranker to understand **relative relevance** within each query group.\n\n---\n\n### 📦 **Output**:\n\n* `rank_X_train`: Feature matrix for ranking\n* `rank_y_train`: Relevance labels (1 for correct match, 0 for others)\n* `train_groups`: List showing the number of candidates per group\n\nThe same is done for the validation set.\n\n---\n\n### 🧠 **Why This Matters**:\n\nThis setup helps the model learn to **rank the correct whale higher** than unrelated ones — a more realistic and effective training objective for identification tasks than plain classification.\n\n\n","metadata":{}},{"cell_type":"code","source":"def create_ranking_data(X, y, group_size=None):\n    \"\"\"\n    Create ranking data for XGBRanker.\n    For whale identification, we create groups where each query is compared against all individuals.\n    \"\"\"\n    n_samples, n_features = X.shape\n    n_classes = len(np.unique(y))\n    \n    if group_size is None:\n        group_size = min(100, n_classes)  # Limit group size for efficiency\n    \n    # Create ranking dataset\n    ranking_X = []\n    ranking_y = []\n    groups = []\n    \n    # For each sample, create a ranking problem\n    for i in range(min(n_samples, 1000)):  # Limit for memory efficiency\n        query_features = X[i]\n        true_label = y[i]\n        \n        # Create group with true label (relevant=1) and random negatives (relevant=0)\n        group_features = [query_features]\n        group_labels = [1]  # True match gets relevance score 1\n        \n        # Add negative samples\n        negative_indices = np.where(y != true_label)[0]\n        if len(negative_indices) > 0:\n            # Sample random negatives\n            n_negatives = min(group_size - 1, len(negative_indices))\n            neg_sample_indices = np.random.choice(negative_indices, n_negatives, replace=False)\n            \n            for neg_idx in neg_sample_indices:\n                group_features.append(X[neg_idx])\n                group_labels.append(0)  # Negative samples get relevance score 0\n        \n        # Add to ranking dataset\n        ranking_X.extend(group_features)\n        ranking_y.extend(group_labels)\n        groups.append(len(group_features))\n    \n    return np.array(ranking_X), np.array(ranking_y), groups\n\n# Create ranking data\nprint(\"Creating ranking dataset for XGBRanker...\")\nrank_X_train, rank_y_train, train_groups = create_ranking_data(X_train, y_train)\nrank_X_val, rank_y_val, val_groups = create_ranking_data(X_val, y_val)\n\nprint(f\"Ranking training set: {rank_X_train.shape[0]} samples, {len(train_groups)} groups\")\nprint(f\"Ranking validation set: {rank_X_val.shape[0]} samples, {len(val_groups)} groups\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n\n---\n\n### 🧠 **What’s Happening:**\n\n#### 1. **Model Setup:**\n\n* An `XGBRanker` model is initialized with:\n\n  * A **pairwise ranking objective** (`rank:pairwise`), which trains the model to prefer relevant items over non-relevant ones within a group.\n  * Common hyperparameters like `max_depth`, `learning_rate`, and `n_estimators` to control model complexity and learning.\n\n---\n\n#### 2. **Training the Model:**\n\n* The model is trained using:\n\n  * Feature matrix (`rank_X_train`)\n  * Relevance labels (`rank_y_train`)\n  * Group sizes (`train_groups`) to tell the model which samples form a ranking problem.\n* A validation set is provided for evaluation during training.\n* The `verbose=10` setting prints progress every 10 iterations.\n\n---\n\n### 📈 **What It Learns:**\n\nInstead of classifying images directly, the model **learns to rank**: given a query image, it should assign **higher scores to correct matches** (same whale ID) than to incorrect ones.\n\nThis is more aligned with real-world tasks like **image-based whale identification**, where **retrieving** the correct match is more important than labeling.\n\n---\n\n","metadata":{}},{"cell_type":"code","source":"# Train XGBRanker\nprint(\"Training XGBRanker...\")\nranker = xgb.XGBRanker(\n    objective='rank:pairwise',\n    n_estimators=100,\n    max_depth=6,\n    learning_rate=0.1,\n    subsample=0.8,\n    colsample_bytree=0.8,\n    random_state=42\n)\n\n# Fit the ranker\nranker.fit(\n    rank_X_train, rank_y_train,\n    group=train_groups,\n    eval_set=[(rank_X_val, rank_y_val)],\n    eval_group=[val_groups],\n    verbose=10\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n---\n\n### 🧪 **Goal**:\n\nTo evaluate how well the trained `XGBRanker` can **retrieve the correct individual whale** given a query image, using **top-5 accuracy** as the metric.\n\n---\n\n### 🔍 **1. Prediction Function (`predict_top_k`)**:\n\n* For a given query image, this function:\n\n  * Compares it against all candidate images (from the training set).\n  * Uses the trained model to compute a **relevance score** for each candidate.\n  * Returns the **top-k most relevant candidate labels and their scores**.\n\n---\n\n### 🧾 **2. Evaluation Loop**:\n\n* For each query image in the validation set:\n\n  * It predicts the top-5 most likely individuals from the training set.\n  * If the true individual is among the top 5 predictions, it's counted as a correct prediction.\n\n---\n\n### ✅ **Final Metric: Top-5 Accuracy**:\n\n* This is the **fraction of queries** for which the correct individual was found in the top 5 predictions.\n* It gives a realistic sense of how useful the model would be in a **retrieval or matching task**, where we care about narrowing down to a few likely candidates.\n\n---\n\n","metadata":{}},{"cell_type":"code","source":"def predict_top_k(ranker, query_features, candidate_features, candidate_labels, k=5):\n    \"\"\"\n    Predict top-k most similar individuals for a query image.\n    \"\"\"\n    # Create feature matrix for ranking\n    n_candidates = len(candidate_features)\n    ranking_features = np.tile(query_features.reshape(1, -1), (n_candidates, 1))\n    \n    # Get ranking scores\n    scores = ranker.predict(ranking_features)\n    \n    # Get top-k predictions\n    top_k_indices = np.argsort(scores)[::-1][:k]\n    top_k_labels = [candidate_labels[i] for i in top_k_indices]\n    top_k_scores = [scores[i] for i in top_k_indices]\n    \n    return top_k_labels, top_k_scores\n\n# Evaluate on validation set\nprint(\"Evaluating model...\")\ncorrect_predictions = 0\ntotal_predictions = 0\n\n# Create candidate set from training data\ncandidate_features = X_train\ncandidate_labels = [label_encoder.inverse_transform([label])[0] for label in y_train]\n\nfor i in range(min(100, len(X_val))):  # Evaluate on subset for speed\n    query_features = X_val[i]\n    true_label = label_encoder.inverse_transform([y_val[i]])[0]\n    \n    # Get top-5 predictions\n    predicted_labels, scores = predict_top_k(\n        ranker, query_features, candidate_features, candidate_labels, k=5\n    )\n    \n    # Check if true label is in top-5\n    if true_label in predicted_labels:\n        correct_predictions += 1\n    total_predictions += 1\n\naccuracy = correct_predictions / total_predictions if total_predictions > 0 else 0\nprint(f\"Top-5 Accuracy: {accuracy:.4f}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n---\n\n### 🧪 **Purpose:**\n\nTo generate **submission-ready predictions** on test images using the trained `XGBRanker`.\n\n---\n\n### 🧾 **What It Does:**\n\n#### 1. **Load Test Images:**\n\n* It goes through all image files in the specified test directory.\n\n#### 2. **Extract Features:**\n\n* For each test image, it extracts HOG features using the same method as during training.\n\n#### 3. **Predict Top-5 Labels:**\n\n* The trained model compares the test image to all training candidates.\n* It returns the **top-5 most likely individual IDs**.\n\n#### 4. **Handle Edge Cases:**\n\n* If fewer than 5 predictions are returned (rare), it fills in with `'new_individual'`.\n* If feature extraction fails, it assigns `'new_individual'` five times as a fallback.\n\n#### 5. **Format for Submission:**\n\n* Each prediction is turned into a string like:\n  `filename, \"whale_001 whale_042 new_individual whale_233 whale_100\"`\n\n---\n\n### 📦 **Output:**\n\nA list of `[filename, prediction_string]` pairs, ready to convert into a DataFrame and save as a submission file.\n\n---\n\n\n","metadata":{}},{"cell_type":"code","source":"# Load test data and make predictions\ndef load_test_images_and_predict(test_dir, ranker, candidate_features, candidate_labels):\n    \"\"\"Load test images and generate predictions for submission.\"\"\"\n    test_predictions = []\n    \n    print(\"Processing test images...\")\n    test_files = [f for f in os.listdir(test_dir) if f.lower().endswith(('.jpg', '.jpeg', '.png'))]\n    \n    for filename in test_files:  # Process subset for demonstration\n        image_path = os.path.join(test_dir, filename)\n        test_features = extract_hog_features(image_path)\n        \n        if test_features is not None:\n            # Get top-5 predictions\n            predicted_labels, scores = predict_top_k(\n                ranker, test_features, candidate_features, candidate_labels, k=5\n            )\n            \n            # Add 'new_individual' if we don't have 5 predictions\n            while len(predicted_labels) < 5:\n                predicted_labels.append('new_individual')\n            \n            # Create prediction string\n            prediction_str = ' '.join(predicted_labels[:5])\n            test_predictions.append([filename, prediction_str])\n        else:\n            # If feature extraction fails, predict new_individual\n            test_predictions.append([filename, 'new_individual ' * 5])\n    \n    return test_predictions","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n\n---\n\n### 📤 **Goal:**\n\nTo create the final **submission file** for the Kaggle competition, containing predicted individual IDs for each test image.\n\n---\n\n### 🧾 **What This Does:**\n\n#### 1. **Run Prediction on Test Images:**\n\n* Calls the `load_test_images_and_predict` function.\n* Uses the trained `XGBRanker` and training set as the candidate gallery.\n* Produces a list of the top-5 predicted IDs for each image.\n\n#### 2. **Create Submission DataFrame:**\n\n* Converts the prediction list into a DataFrame with two columns:\n\n  * `'image'`: the filename\n  * `'predictions'`: a space-separated string of the top-5 predicted IDs\n\n#### 3. **Save Submission File:**\n\n* Exports the DataFrame to `submission.csv` in the required format for submission to Kaggle.\n\n---\n\n","metadata":{}},{"cell_type":"code","source":"# Generate test predictions (uncomment when ready to process test set)\ntest_dir = '/kaggle/input/happywhale-cropped-removebackground-v1/removedBackground_test_image'\ntest_predictions = load_test_images_and_predict(test_dir, ranker, candidate_features, candidate_labels)\n\n# Create submission file\nsubmission_df = pd.DataFrame(test_predictions, columns=['image', 'predictions'])\ndisplay(submission_df)\nprint(len(submission_df))\nsubmission_df.to_csv('submission.csv', index=False)\nprint(\"Submission file created: submission.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Overall Pipline\n\n---\n\n## 🐋 **Whale Identification Pipeline Overview**\n\n---\n\n### **1. Preprocessing & Feature Extraction**\n\n* **Input**: Cropped and background-removed whale images.\n* **HOG Extraction**:\n\n  * Convert each image to grayscale.\n  * Resize to a standard size (e.g. 128×128).\n  * Extract **Histogram of Oriented Gradients (HOG)** features to represent shape and edge structure.\n\n---\n\n### **2. Label Encoding**\n\n* Convert whale `individual_id` (strings) into numeric labels using `LabelEncoder`.\n\n---\n\n### **3. Train/Validation Split**\n\n* Stratified split of feature vectors and labels into **training and validation sets**.\n* Maintains class distribution to ensure balanced evaluation.\n\n---\n\n### **4. Ranking Data Construction**\n\n* For each query image:\n\n  * Construct a group: 1 **relevant** sample (same ID) + multiple **irrelevant** samples (different IDs).\n  * Assign **relevance scores**: `1` for true match, `0` for others.\n  * Build group metadata to inform the ranker about the structure of each ranking problem.\n\n---\n\n### **5. Train XGBRanker**\n\n* Use `XGBRanker` with `rank:pairwise` objective.\n* Fit on training ranking data with group sizes.\n* Validate on a held-out set using group-based evaluation.\n\n---\n\n### **6. Evaluation (Top-k Accuracy)**\n\n* For each validation sample:\n\n  * Predict top-5 most similar individuals from training candidates.\n  * Check if true label appears in the top-5.\n* Report **Top-5 Accuracy** as the performance metric.\n\n---\n\n### **7. Test Prediction**\n\n* Load and process test images the same way (HOG).\n* Predict top-5 most similar individuals using trained ranker.\n* If no match, fill with `'new_individual'`.\n\n---\n\n### **8. Submission File Creation**\n\n* Format predictions as a space-separated string of top-5 IDs.\n* Save as a CSV file with columns:\n\n\n---\n","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}