{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":124685,"databundleVersionId":14664296,"sourceType":"competition"}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"57ac069b-2324-4d8c-b11f-778868373859","_cell_guid":"ff5219d2-3be9-499b-8790-b751d59b3194","trusted":true,"collapsed":true,"jupyter":{"outputs_hidden":true},"execution":{"iopub.status.busy":"2026-02-15T11:28:51.840227Z","iopub.execute_input":"2026-02-15T11:28:51.840442Z","iopub.status.idle":"2026-02-15T11:28:54.951912Z","shell.execute_reply.started":"2026-02-15T11:28:51.840421Z","shell.execute_reply":"2026-02-15T11:28:54.951107Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda.amp import autocast\nimport torchvision.transforms as T\nfrom torchvision.transforms.functional import hflip, vflip\n\nfrom PIL import Image\nfrom tqdm.auto import tqdm\nfrom transformers import AutoImageProcessor, AutoModelForImageClassification\nfrom sklearn.preprocessing import normalize\nimport csv\nfrom typing import List, Tuple, Optional, Dict\nimport json\nfrom pathlib import Path","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T11:28:54.953902Z","iopub.execute_input":"2026-02-15T11:28:54.954459Z","iopub.status.idle":"2026-02-15T11:29:23.387534Z","shell.execute_reply.started":"2026-02-15T11:28:54.954422Z","shell.execute_reply":"2026-02-15T11:29:23.386929Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Enhanced Configuration","metadata":{}},{"cell_type":"code","source":"\nclass Config:\n    \"\"\"Centralized configuration management\"\"\"\n    \n    def __init__(self):\n        self.device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n        self.batch_size = 8 if torch.cuda.is_available() else 2\n        self.num_workers = 4\n        self.pin_memory = True if torch.cuda.is_available() else False\n        \n        # Model settings\n        self.model_name = \"gerald29/plantclef2024\"\n        self.mixed_precision = True\n        \n        # Inference settings\n        self.tta_transforms = ['original', 'hflip', 'vflip', 'rotate90']\n        self.temperature_grid = [0.5, 0.7, 0.8, 1.0, 1.2, 1.5, 2.0]\n        self.target_entropy = 0.6\n        \n        # Threshold settings\n        self.threshold_grid_start = 0.02\n        self.threshold_grid_end = 0.50\n        self.threshold_grid_step = 0.02\n        self.target_avg_species = 4.5\n        \n        # Dynamic K settings\n        self.min_species = 1\n        self.max_species = 10\n        self.confidence_threshold = 0.15\n        \n        # Paths\n        self.data_root = \"/kaggle/input/plantclef-2026\"\n        self.test_img_dir = f\"{self.data_root}/PlantCLEF2025_test_images/PlantCLEF2025_test_images\"\n        self.test_csv = f\"{self.data_root}/PlantCLEF2025_test.csv\"\n        self.species_csv = f\"{self.data_root}/species_ids.csv\"\n        \n        # Optimization\n        self.use_cache = True\n        self.cache_dir = \"/kaggle/temp/cache\"\n\nCFG = Config()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T11:29:23.388367Z","iopub.execute_input":"2026-02-15T11:29:23.388863Z","iopub.status.idle":"2026-02-15T11:29:23.395795Z","shell.execute_reply.started":"2026-02-15T11:29:23.388827Z","shell.execute_reply":"2026-02-15T11:29:23.395218Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Environment Setup","metadata":{}},{"cell_type":"code","source":"def setup_environment():\n    \"\"\"Setup environment and check resources\"\"\"\n    # Create cache directory\n    if CFG.use_cache:\n        os.makedirs(CFG.cache_dir, exist_ok=True)\n    \n    # Check GPU memory\n    if torch.cuda.is_available():\n        gpu_memory = torch.cuda.get_device_properties(0).total_memory / 1e9\n        print(f\"GPU: {torch.cuda.get_device_name(0)}\")\n        print(f\"GPU Memory: {gpu_memory:.2f} GB\")\n        \n        # Adjust batch size based on GPU memory\n        if gpu_memory < 8:\n            CFG.batch_size = min(CFG.batch_size, 4)\n            print(f\"Adjusted batch size to {CFG.batch_size} due to GPU memory constraints\")\n    \n    # Check internet connection\n    import socket\n    try:\n        socket.create_connection((\"huggingface.co\", 80), timeout=5)\n        print(\"✓ Internet connection available\")\n    except:\n        print(\"✗ No internet connection - using local models only\")\n    \n    return True\n\nsetup_environment()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T11:29:23.396853Z","iopub.execute_input":"2026-02-15T11:29:23.397137Z","iopub.status.idle":"2026-02-15T11:29:23.447184Z","shell.execute_reply.started":"2026-02-15T11:29:23.397107Z","shell.execute_reply":"2026-02-15T11:29:23.446624Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DataManager:\n    \"\"\"Centralized data management\"\"\"\n    \n    def __init__(self, cfg: Config):\n        self.cfg = cfg\n        self.test_df = None\n        self.species_df = None\n        self.species_ids = None\n        self.species_to_idx = {}\n        self.idx_to_species = {}\n    \n    def load_data(self):\n        \"\"\"Load all necessary data files\"\"\"\n        print(\"Loading data files...\")\n        \n        # Load test data\n        self.test_df = pd.read_csv(\n            self.cfg.test_csv, \n            sep=\";\", \n            engine=\"python\"\n        )\n        self.test_df.columns = self.test_df.columns.str.strip().str.strip('\"')\n        \n        # Validate image paths\n        self.test_df = self._validate_image_paths(self.test_df)\n        \n        # Load species mapping\n        self.species_df = pd.read_csv(self.cfg.species_csv)\n        self.species_ids = self.species_df[\"species_id\"].values\n        \n        # Create mappings\n        for idx, species_id in enumerate(self.species_ids):\n            self.species_to_idx[species_id] = idx\n            self.idx_to_species[idx] = species_id\n        \n        print(f\"✓ Loaded {len(self.test_df)} test images\")\n        print(f\"✓ Total species: {len(self.species_ids)}\")\n        \n        return self.test_df, self.species_ids\n    \n    def _validate_image_paths(self, df):\n        \"\"\"Validate that image files exist\"\"\"\n        valid_rows = []\n        for idx, row in df.iterrows():\n            img_path = os.path.join(\n                self.cfg.test_img_dir, \n                f\"{row['quadrat_id']}.jpg\"\n            )\n            if os.path.exists(img_path):\n                valid_rows.append(idx)\n            else:\n                print(f\"Warning: Image not found - {img_path}\")\n        \n        return df.loc[valid_rows].reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T11:29:23.447982Z","iopub.execute_input":"2026-02-15T11:29:23.448305Z","iopub.status.idle":"2026-02-15T11:29:23.455431Z","shell.execute_reply.started":"2026-02-15T11:29:23.448274Z","shell.execute_reply":"2026-02-15T11:29:23.454729Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Quardrat Enhancement","metadata":{}},{"cell_type":"code","source":"class EnhancedQuadratDataset(Dataset):\n    \"\"\"Enhanced dataset with caching and augmentation support\"\"\"\n    \n    def __init__(self, df, img_dir, processor, transform=None, use_cache=False):\n        self.df = df\n        self.img_dir = img_dir\n        self.processor = processor\n        self.transform = transform\n        self.use_cache = use_cache\n        self.cache = {}\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        quadrat_id = row[\"quadrat_id\"]\n        \n        # Check cache\n        if self.use_cache and quadrat_id in self.cache:\n            return self.cache[quadrat_id], quadrat_id\n        \n        img_path = os.path.join(self.img_dir, f\"{quadrat_id}.jpg\")\n        \n        try:\n            # Load and process image\n            image = Image.open(img_path).convert(\"RGB\")\n            \n            # Apply custom transforms if provided\n            if self.transform:\n                image = self.transform(image)\n            \n            # Process with model processor\n            inputs = self.processor(\n                images=image,\n                return_tensors=\"pt\"\n            )\n            \n            pixel_values = inputs[\"pixel_values\"].squeeze(0)\n            \n            # Cache if enabled\n            if self.use_cache:\n                self.cache[quadrat_id] = pixel_values\n            \n            return pixel_values, quadrat_id\n            \n        except Exception as e:\n            print(f\"Error loading image {img_path}: {e}\")\n            # Return a black image as fallback\n            dummy = torch.zeros((3, 518, 518))\n            return dummy, quadrat_id","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T11:29:23.456225Z","iopub.execute_input":"2026-02-15T11:29:23.456459Z","iopub.status.idle":"2026-02-15T11:29:23.470670Z","shell.execute_reply.started":"2026-02-15T11:29:23.456439Z","shell.execute_reply":"2026-02-15T11:29:23.469872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ModelManager:\n    \"\"\"Model management with optimization features\"\"\"\n    \n    def __init__(self, cfg: Config):\n        self.cfg = cfg\n        self.model = None\n        self.processor = None\n        \n    def load_model(self):\n        \"\"\"Load and optimize model\"\"\"\n        print(\"Loading model...\")\n        \n        # Load processor and model\n        self.processor = AutoImageProcessor.from_pretrained(self.cfg.model_name)\n        self.model = AutoModelForImageClassification.from_pretrained(self.cfg.model_name)\n        \n        # Move to device and set precision\n        self.model = self.model.to(self.cfg.device)\n        \n        # Optimization: Use float16 if GPU available\n        if self.cfg.device == \"cuda\" and self.cfg.mixed_precision:\n            self.model = self.model.half()\n            print(\"✓ Model loaded in FP16 precision\")\n        else:\n            self.model = self.model.float()\n            print(\"✓ Model loaded in FP32 precision\")\n        \n        # Set to eval mode\n        self.model.eval()\n        \n        # Compile model for faster inference (PyTorch 2.0+)\n        if hasattr(torch, 'compile'):\n            try:\n                self.model = torch.compile(self.model, mode=\"reduce-overhead\")\n                print(\"✓ Model compiled with torch.compile\")\n            except:\n                print(\"⚠ torch.compile not available\")\n        \n        return self.model, self.processor","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T11:29:23.472756Z","iopub.execute_input":"2026-02-15T11:29:23.472983Z","iopub.status.idle":"2026-02-15T11:29:23.484936Z","shell.execute_reply.started":"2026-02-15T11:29:23.472963Z","shell.execute_reply":"2026-02-15T11:29:23.484245Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# TTA Inferences","metadata":{}},{"cell_type":"code","source":"class TTAInference:\n    \"\"\"Test Time Augmentation with multiple strategies\"\"\"\n    \n    def __init__(self, model, device, transforms=['original', 'hflip']):\n        self.model = model\n        self.device = device\n        self.transforms = transforms\n        \n    @torch.no_grad()\n    def predict_batch(self, pixel_values):\n        \"\"\"Apply TTA to a batch\"\"\"\n        all_logits = []\n        \n        for transform in self.transforms:\n            if transform == 'original':\n                augmented = pixel_values\n            elif transform == 'hflip':\n                augmented = torch.flip(pixel_values, dims=[-1])\n            elif transform == 'vflip':\n                augmented = torch.flip(pixel_values, dims=[-2])\n            elif transform == 'rotate90':\n                augmented = torch.rot90(pixel_values, k=1, dims=[-2, -1])\n            elif transform == 'rotate270':\n                augmented = torch.rot90(pixel_values, k=3, dims=[-2, -1])\n            else:\n                augmented = pixel_values\n            \n            # Mixed precision inference\n            if CFG.mixed_precision and CFG.device == \"cuda\":\n                with autocast():\n                    outputs = self.model(pixel_values=augmented)\n                    logits = outputs.logits.float()\n            else:\n                outputs = self.model(pixel_values=augmented)\n                logits = outputs.logits\n            \n            all_logits.append(logits)\n        \n        # Average predictions\n        final_logits = torch.stack(all_logits).mean(dim=0)\n        return final_logits","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T11:29:23.485818Z","iopub.execute_input":"2026-02-15T11:29:23.486530Z","iopub.status.idle":"2026-02-15T11:29:23.498968Z","shell.execute_reply.started":"2026-02-15T11:29:23.486498Z","shell.execute_reply":"2026-02-15T11:29:23.498327Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Advance Inferences","metadata":{}},{"cell_type":"code","source":"class AdvancedInference:\n    \"\"\"Advanced inference pipeline with optimizations\"\"\"\n    \n    def __init__(self, model, processor, cfg: Config):\n        self.model = model\n        self.processor = processor\n        self.cfg = cfg\n        self.tta = TTAInference(model, cfg.device, cfg.tta_transforms)\n        \n    def run_inference(self, dataloader, temperature=1.0, desc=\"Inference\"):\n        \"\"\"Run inference with progress bar and optimizations\"\"\"\n        all_logits = []\n        all_ids = []\n        \n        self.model.eval()\n        \n        with torch.no_grad():\n            for batch_idx, (pixel_values, ids) in enumerate(tqdm(dataloader, desc=desc)):\n                # Move to device with proper dtype\n                if self.cfg.mixed_precision and self.cfg.device == \"cuda\":\n                    pixel_values = pixel_values.to(self.cfg.device).half()\n                else:\n                    pixel_values = pixel_values.to(self.cfg.device).float()\n                \n                # Get predictions with TTA\n                logits = self.tta.predict_batch(pixel_values)\n                \n                # Apply temperature scaling\n                logits = logits / temperature\n                \n                all_logits.append(logits.cpu())\n                all_ids.extend(ids)\n                \n                # Memory management\n                if batch_idx % 50 == 0:\n                    torch.cuda.empty_cache() if torch.cuda.is_available() else None\n        \n        all_logits = torch.cat(all_logits, dim=0)\n        return all_logits, all_ids\n    \n    def auto_tune_temperature(self, dataloader, target_entropy=0.6):\n        \"\"\"Automatically tune temperature based on entropy\"\"\"\n        print(\"Auto-tuning temperature...\")\n        \n        best_temperature = 1.0\n        best_diff = float('inf')\n        \n        for temp in self.cfg.temperature_grid:\n            logits, _ = self.run_inference(\n                dataloader, \n                temperature=temp,\n                desc=f\"T={temp:.1f}\"\n            )\n            \n            probs = torch.softmax(logits, dim=1).numpy()\n            \n            # Calculate entropy\n            entropies = -np.sum(probs * np.log(probs + 1e-12), axis=1)\n            mean_entropy = np.mean(entropies)\n            normalized_entropy = mean_entropy / np.log(probs.shape[1])\n            \n            diff = abs(normalized_entropy - target_entropy)\n            \n            print(f\"  T={temp:.1f} -> Entropy: {normalized_entropy:.4f}\")\n            \n            if diff < best_diff:\n                best_diff = diff\n                best_temperature = temp\n        \n        print(f\"✓ Selected temperature: {best_temperature}\")\n        return best_temperature","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T11:29:23.500501Z","iopub.execute_input":"2026-02-15T11:29:23.500731Z","iopub.status.idle":"2026-02-15T11:29:23.513480Z","shell.execute_reply.started":"2026-02-15T11:29:23.500710Z","shell.execute_reply":"2026-02-15T11:29:23.512782Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Probability Refiner","metadata":{}},{"cell_type":"code","source":"class ProbabilityRefiner:\n    \"\"\"Advanced probability refinement strategies\"\"\"\n    \n    @staticmethod\n    def refine_probabilities(probs, strategy='adaptive'):\n        \"\"\"Refine probability distribution\"\"\"\n        \n        if strategy == 'adaptive':\n            return ProbabilityRefiner._adaptive_refinement(probs)\n        elif strategy == 'entropy_based':\n            return ProbabilityRefiner._entropy_based_refinement(probs)\n        elif strategy == 'top_k_smooth':\n            return ProbabilityRefiner._top_k_smoothing(probs)\n        else:\n            return probs\n    \n    @staticmethod\n    def _adaptive_refinement(probs):\n        \"\"\"Adaptive refinement based on confidence\"\"\"\n        refined = probs.copy()\n        max_conf = probs.max()\n        entropy = -np.sum(probs * np.log(probs + 1e-12))\n        \n        # High confidence: sharpen\n        if max_conf > 0.7:\n            power = 1.5 + (max_conf - 0.7) * 2  # Dynamic sharpening\n            refined = refined ** power\n        # Medium confidence: mild adjustment\n        elif max_conf > 0.3:\n            refined = refined ** 1.1\n        # Low confidence: smooth\n        else:\n            power = 0.8 - (0.3 - max_conf) * 0.5\n            refined = refined ** max(0.5, power)\n        \n        # Re-normalize\n        refined = refined / (refined.sum() + 1e-12)\n        \n        return refined\n    \n    @staticmethod\n    def _entropy_based_refinement(probs):\n        \"\"\"Entropy-based refinement\"\"\"\n        entropy = -np.sum(probs * np.log(probs + 1e-12))\n        max_entropy = np.log(len(probs))\n        normalized_entropy = entropy / max_entropy\n        \n        # Adjust based on entropy\n        if normalized_entropy < 0.3:  # Low entropy = high confidence\n            refined = probs ** 1.5\n        elif normalized_entropy > 0.7:  # High entropy = low confidence\n            refined = probs ** 0.7\n        else:\n            refined = probs\n        \n        return refined / (refined.sum() + 1e-12)\n    \n    @staticmethod\n    def _top_k_smoothing(probs, k=10):\n        \"\"\"Smooth probabilities keeping top-k\"\"\"\n        top_k_idx = np.argsort(probs)[-k:]\n        refined = np.zeros_like(probs)\n        refined[top_k_idx] = probs[top_k_idx]\n        \n        # Apply smoothing to top-k\n        refined[top_k_idx] = np.power(refined[top_k_idx], 0.9)\n        \n        return refined / (refined.sum() + 1e-12)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T11:29:23.514187Z","iopub.execute_input":"2026-02-15T11:29:23.514433Z","iopub.status.idle":"2026-02-15T11:29:23.527517Z","shell.execute_reply.started":"2026-02-15T11:29:23.514400Z","shell.execute_reply":"2026-02-15T11:29:23.526860Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Optimizer","metadata":{}},{"cell_type":"code","source":"class ThresholdOptimizer:\n    \"\"\"Optimize threshold selection\"\"\"\n    \n    def __init__(self, cfg: Config):\n        self.cfg = cfg\n        \n    def grid_search_threshold(self, probs_array, target_avg=4.5):\n        \"\"\"Grid search for optimal threshold\"\"\"\n        \n        thresholds = np.arange(\n            self.cfg.threshold_grid_start,\n            self.cfg.threshold_grid_end,\n            self.cfg.threshold_grid_step\n        )\n        \n        results = []\n        \n        for threshold in thresholds:\n            species_counts = []\n            \n            for probs in probs_array:\n                selected = np.where(probs >= threshold)[0]\n                species_counts.append(len(selected))\n            \n            avg_species = np.mean(species_counts)\n            std_species = np.std(species_counts)\n            \n            # Calculate score (minimize distance to target and std)\n            score = abs(avg_species - target_avg) + std_species * 0.1\n            \n            results.append({\n                'threshold': threshold,\n                'avg_species': avg_species,\n                'std_species': std_species,\n                'score': score\n            })\n        \n        # Find best threshold\n        results_df = pd.DataFrame(results)\n        best_idx = results_df['score'].idxmin()\n        best_result = results_df.iloc[best_idx]\n        \n        print(f\"✓ Optimal threshold: {best_result['threshold']:.3f}\")\n        print(f\"  Avg species: {best_result['avg_species']:.2f} ± {best_result['std_species']:.2f}\")\n        \n        return best_result['threshold'], results_df\n    \n    def adaptive_threshold(self, probs, base_threshold, entropy_weight=0.3):\n        \"\"\"Compute adaptive threshold per sample\"\"\"\n        \n        # Calculate entropy\n        entropy = -np.sum(probs * np.log(probs + 1e-12))\n        normalized_entropy = entropy / np.log(len(probs))\n        \n        # Adjust threshold based on entropy\n        # High entropy -> lower threshold (more uncertain, include more)\n        # Low entropy -> higher threshold (more certain, be selective)\n        adaptive_thresh = base_threshold * (1 - entropy_weight * (normalized_entropy - 0.5))\n        \n        return np.clip(adaptive_thresh, 0.01, 0.9)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T11:29:23.528324Z","iopub.execute_input":"2026-02-15T11:29:23.528593Z","iopub.status.idle":"2026-02-15T11:29:23.541358Z","shell.execute_reply.started":"2026-02-15T11:29:23.528564Z","shell.execute_reply":"2026-02-15T11:29:23.540636Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submission Generation","metadata":{}},{"cell_type":"code","source":"class SubmissionGenerator:\n    \"\"\"Generate submission with advanced strategies\"\"\"\n    \n    def __init__(self, cfg: Config, refiner: ProbabilityRefiner, optimizer: ThresholdOptimizer):\n        self.cfg = cfg\n        self.refiner = refiner\n        self.optimizer = optimizer\n        \n    def generate_predictions(self, probs_array, ids_array, species_ids, base_threshold):\n        \"\"\"Generate final predictions\"\"\"\n        \n        submission_rows = []\n        stats = {'total': 0, 'avg_species': 0, 'min_species': 999, 'max_species': 0}\n        \n        for i, (probs, quadrat_id) in enumerate(zip(probs_array, ids_array)):\n            # Refine probabilities\n            refined_probs = self.refiner.refine_probabilities(probs, strategy='adaptive')\n            \n            # Get adaptive threshold\n            adaptive_thresh = self.optimizer.adaptive_threshold(\n                refined_probs, \n                base_threshold,\n                entropy_weight=0.3\n            )\n            \n            # Dynamic K based on entropy\n            entropy = -np.sum(refined_probs * np.log(refined_probs + 1e-12))\n            normalized_entropy = entropy / np.log(len(refined_probs))\n            \n            dynamic_k = int(\n                self.cfg.min_species + \n                normalized_entropy * (self.cfg.max_species - self.cfg.min_species)\n            )\n            dynamic_k = np.clip(dynamic_k, self.cfg.min_species, self.cfg.max_species)\n            \n            # Select species\n            selected_indices = self._select_species(\n                refined_probs, \n                adaptive_thresh, \n                dynamic_k\n            )\n            \n            # Map to species IDs\n            predicted_species = species_ids[selected_indices]\n            species_list_str = \"[\" + \", \".join(map(str, predicted_species)) + \"]\"\n            \n            submission_rows.append({\n                \"quadrat_id\": quadrat_id,\n                \"species_ids\": species_list_str\n            })\n            \n            # Update statistics\n            num_species = len(selected_indices)\n            stats['total'] += num_species\n            stats['min_species'] = min(stats['min_species'], num_species)\n            stats['max_species'] = max(stats['max_species'], num_species)\n        \n        stats['avg_species'] = stats['total'] / len(probs_array)\n        \n        return pd.DataFrame(submission_rows), stats\n    \n    def _select_species(self, probs, threshold, max_k):\n        \"\"\"Select species based on threshold and max_k\"\"\"\n        \n        # Primary selection by threshold\n        selected = np.where(probs >= threshold)[0]\n        \n        # If too many, keep top-k\n        if len(selected) > max_k:\n            top_k_indices = np.argsort(probs)[-max_k:]\n            selected = top_k_indices\n        \n        # If none selected, take at least top-1\n        if len(selected) == 0:\n            selected = [np.argmax(probs)]\n        \n        # Additional filtering: remove very low confidence predictions\n        final_selected = []\n        for idx in selected:\n            if probs[idx] >= self.cfg.confidence_threshold or len(final_selected) == 0:\n                final_selected.append(idx)\n        \n        return np.array(final_selected)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T11:29:23.542213Z","iopub.execute_input":"2026-02-15T11:29:23.542519Z","iopub.status.idle":"2026-02-15T11:29:23.554877Z","shell.execute_reply.started":"2026-02-15T11:29:23.542487Z","shell.execute_reply":"2026-02-15T11:29:23.554182Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Pipeline Assembly","metadata":{}},{"cell_type":"code","source":"def main_pipeline():\n    \"\"\"Main execution pipeline\"\"\"\n    \n    print(\"=\"*50)\n    print(\"PlantCLEF 2026 - Enhanced Pipeline\")\n    print(\"=\"*50)\n    \n    # Initialize components\n    data_manager = DataManager(CFG)\n    model_manager = ModelManager(CFG)\n    \n    # Load data\n    test_df, species_ids = data_manager.load_data()\n    \n    # Load model\n    model, processor = model_manager.load_model()\n    \n    # Create dataset and dataloader\n    test_dataset = EnhancedQuadratDataset(\n        test_df,\n        CFG.test_img_dir,\n        processor,\n        use_cache=CFG.use_cache\n    )\n    \n    test_loader = DataLoader(\n        test_dataset,\n        batch_size=CFG.batch_size,\n        shuffle=False,\n        num_workers=CFG.num_workers,\n        pin_memory=CFG.pin_memory,\n        drop_last=False\n    )\n    \n    # Initialize inference\n    inference = AdvancedInference(model, processor, CFG)\n    \n    # Auto-tune temperature\n    best_temperature = inference.auto_tune_temperature(\n        test_loader, \n        target_entropy=CFG.target_entropy\n    )\n    \n    # Run final inference\n    print(\"\\nRunning final inference...\")\n    final_logits, all_ids = inference.run_inference(\n        test_loader,\n        temperature=best_temperature,\n        desc=\"Final Inference\"\n    )\n    \n    # Convert to probabilities\n    final_probs = torch.softmax(final_logits, dim=1).numpy()\n    \n    # Initialize optimization components\n    refiner = ProbabilityRefiner()\n    optimizer = ThresholdOptimizer(CFG)\n    \n    # Find optimal threshold\n    best_threshold, threshold_results = optimizer.grid_search_threshold(\n        final_probs,\n        target_avg=CFG.target_avg_species\n    )\n    \n    # Generate submission\n    generator = SubmissionGenerator(CFG, refiner, optimizer)\n    submission_df, stats = generator.generate_predictions(\n        final_probs,\n        all_ids,\n        species_ids,\n        best_threshold\n    )\n    \n    # Print statistics\n    print(\"\\n\" + \"=\"*50)\n    print(\"Submission Statistics:\")\n    print(f\"  Average species per quadrat: {stats['avg_species']:.2f}\")\n    print(f\"  Min species: {stats['min_species']}\")\n    print(f\"  Max species: {stats['max_species']}\")\n    print(\"=\"*50)\n    \n    # Save submission\n    submission_df.to_csv(\n        \"submission_enhanced.csv\",\n        sep=\",\",\n        index=False,\n        quoting=csv.QUOTE_ALL\n    )\n    \n    print(\"\\n✓ Submission saved to 'submission_enhanced.csv'\")\n    \n    # Save configuration and statistics\n    config_log = {\n        'temperature': float(best_temperature),\n        'threshold': float(best_threshold),\n        'statistics': stats,\n        'tta_transforms': CFG.tta_transforms\n    }\n    \n    with open('inference_log.json', 'w') as f:\n        json.dump(config_log, f, indent=2)\n    \n    print(\"✓ Configuration saved to 'inference_log.json'\")\n    \n    return submission_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T11:29:23.556400Z","iopub.execute_input":"2026-02-15T11:29:23.556648Z","iopub.status.idle":"2026-02-15T11:29:23.569663Z","shell.execute_reply.started":"2026-02-15T11:29:23.556629Z","shell.execute_reply":"2026-02-15T11:29:23.568951Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Main Run","metadata":{}},{"cell_type":"code","source":"if __name__ == \"__main__\":\n    submission_df = main_pipeline()\n    \n    # Display sample predictions\n    print(\"\\nSample predictions:\")\n    print(submission_df.head(10))\n    \n    # Clean up\n    gc.collect()\n    torch.cuda.empty_cache() if torch.cuda.is_available() else None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-02-15T11:29:23.570601Z","iopub.execute_input":"2026-02-15T11:29:23.571059Z","iopub.status.idle":"2026-02-15T12:05:19.237232Z","shell.execute_reply.started":"2026-02-15T11:29:23.571037Z","shell.execute_reply":"2026-02-15T12:05:19.236210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}