{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11053663,"sourceType":"datasetVersion","datasetId":6886569},{"sourceId":404336,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":330512,"modelId":351343}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport logging\nimport random\nimport gc\nimport time\nimport cv2\nimport math\nimport warnings\nfrom pathlib import Path\nfrom functools import partial\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport librosa\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda.amp import autocast, GradScaler\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm.auto import tqdm\n\nimport timm\n\nwarnings.filterwarnings(\"ignore\")\nlogging.basicConfig(level=logging.ERROR)\nclass CFG:\n    \n    seed = 42\n    debug = True  \n    apex = False\n    print_freq = 100\n    num_workers = 4  # Increased from 2\n    \n    # Detect environment\n    # Check if we're in Kaggle\n    if os.path.exists('/kaggle/input'):\n        print(\"Running in Kaggle environment\")\n        is_kaggle = True\n        BASE_PATH = '/kaggle/input/birdclef-2025'\n    else:\n        print(\"Running in local environment\")\n        is_kaggle = False\n        # Look for the data in the current directory or parent directory\n        if os.path.exists('./train.csv'):\n            BASE_PATH = '.'\n        elif os.path.exists('../train.csv'):\n            BASE_PATH = '..'\n        else:\n            BASE_PATH = './data'  # Default fallback\n    \n    OUTPUT_DIR = '/kaggle/working/' if is_kaggle else './outputs'\n    \n    # Create output directory if it doesn't exist\n    os.makedirs(OUTPUT_DIR, exist_ok=True)\n\n    train_datadir = f'{BASE_PATH}/train_audio'\n    train_csv = f'{BASE_PATH}/train.csv'\n    test_soundscapes = f'{BASE_PATH}/test_soundscapes'\n    submission_csv = f'{BASE_PATH}/sample_submission.csv'\n    taxonomy_csv = f'{BASE_PATH}/taxonomy.csv'\n    \n    spectrogram_npy = '/kaggle/input/birdclef25-mel-spectrograms/birdclef2025_melspec_5sec_256_256.npy' if is_kaggle else None\n    \n    model_name = 'mobilenetv3_small_100'  # Changed from efficientnet_b3\n    pretrained = False  # Changed to False for Kaggle (offline usage)\n    pretrained_weights = None  # Path to local weights file, set this if you have downloaded weights\n    in_channels = 1\n    \n    LOAD_DATA = True  \n    USE_AMP = True  # Enable mixed precision\n    PIN_MEMORY = True  # Pin memory for faster data loading\n    \n    FS = 32000\n    TARGET_DURATION = 5.0\n    TARGET_SHAPE = (256, 256)\n    \n    N_FFT = 1024\n    HOP_LENGTH = 512\n    N_MELS = 128\n    FMIN = 50\n    FMAX = 14000\n    \n    device = 'cuda' if torch.cuda.is_available() else 'cpu'\n    epochs = 15  # Increased from 10\n    batch_size = 64  # Increased for MobileNetV3 which is smaller than EfficientNet\n    gradient_accumulation_steps = 1  # Reduced since MobileNetV3 is more memory efficient\n    criterion = 'BCEWithLogitsLoss'\n\n    n_fold = 5\n    selected_folds = [0, 1, 2, 3, 4]   \n\n    optimizer = 'AdamW'\n    lr = 2e-4  # Slightly higher learning rate for MobileNetV3 which converges faster\n    weight_decay = 5e-5  # Reduced for MobileNetV3 to prevent overfitting\n  \n    scheduler = 'CosineAnnealingWarmRestarts'  # Changed from CosineAnnealingLR\n    min_lr = 1e-6\n    T_0 = 5  # For CosineAnnealingWarmRestarts\n    T_mult = 1  # For CosineAnnealingWarmRestarts\n\n    aug_prob = 0.7  # Increased from 0.5\n    mixup_alpha = 0.4\n    cutmix_alpha = 0.4  # Added cutmix\n    \n    def update_debug_settings(self):\n        if self.debug:\n            self.epochs = 2\n            self.selected_folds = [0]\n\ncfg = CFG()\ndef set_seed(seed=42):\n    \"\"\"\n    Set seed for reproducibility\n    \"\"\"\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n\nset_seed(cfg.seed)\n\n# Memory-efficient spectrograms loading function\ndef load_spectrograms_mmap(file_path):\n    \"\"\"Load spectrograms with optional memory mapping and chunking for large datasets\"\"\"\n    if not os.path.exists(file_path):\n        print(f\"Warning: Spectrogram file not found at {file_path}\")\n        return {}\n        \n    try:\n        print(f\"Loading spectrograms from {file_path}\")\n        # First try to load with memory mapping\n        return np.load(file_path, allow_pickle=True, mmap_mode='r').item()\n    except ValueError as e:\n        print(f\"Memory mapping failed: {e}. Loading entire file into memory instead.\")\n        try:\n            # Try loading everything at once\n            data = np.load(file_path, allow_pickle=True).item()\n            print(f\"Successfully loaded {len(data)} spectrograms into memory\")\n            return data\n        except MemoryError:\n            # If we hit memory error, try to load in chunks\n            print(\"Memory error when loading spectrograms. Attempting to load in chunks...\")\n            try:\n                # Load just the metadata first\n                data = {}\n                with np.load(file_path, allow_pickle=True) as loaded:\n                    # Process only a subset of data if in debug mode\n                    if cfg.debug:\n                        keys = list(loaded.item().keys())[:1000]\n                        print(f\"Debug mode: Loading only {len(keys)} spectrograms\")\n                        for k in keys:\n                            data[k] = loaded.item()[k]\n                    else:\n                        # Try to process in smaller chunks\n                        all_data = loaded.item()\n                        keys = list(all_data.keys())\n                        chunk_size = 1000\n                        for i in range(0, len(keys), chunk_size):\n                            chunk_keys = keys[i:i+chunk_size]\n                            print(f\"Loading chunk {i//chunk_size + 1}/{(len(keys)-1)//chunk_size + 1}...\")\n                            for k in chunk_keys:\n                                data[k] = all_data[k]\n                            # Force garbage collection\n                            gc.collect()\n                print(f\"Successfully loaded {len(data)} spectrograms in chunks\")\n                return data\n            except Exception as e2:\n                print(f\"Failed to load spectrograms even in chunks: {e2}\")\n                print(\"Continuing without pre-computed spectrograms\")\n                return {}\n\nclass BirdCLEFDatasetFromNPY(Dataset):\n    def __init__(self, df, cfg, spectrograms=None, mode=\"train\"):\n        self.df = df\n        self.cfg = cfg\n        self.mode = mode\n        self.spectrograms = spectrograms\n        \n        taxonomy_df = pd.read_csv(self.cfg.taxonomy_csv)\n        self.species_ids = taxonomy_df['primary_label'].tolist()\n        self.num_classes = len(self.species_ids)\n        self.label_to_idx = {label: idx for idx, label in enumerate(self.species_ids)}\n\n        if 'filepath' not in self.df.columns:\n            self.df['filepath'] = self.cfg.train_datadir + '/' + self.df.filename\n        \n        if 'samplename' not in self.df.columns:\n            self.df['samplename'] = self.df.filename.map(lambda x: x.split('/')[0] + '-' + x.split('/')[-1].split('.')[0])\n\n        self.sample_names = self.df['samplename'].values\n        \n        if cfg.debug:\n            self.df = self.df.sample(min(1000, len(self.df)), random_state=cfg.seed).reset_index(drop=True)\n            self.sample_names = self.df['samplename'].values\n    \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        samplename = row['samplename']\n        \n        # Faster lookup\n        if self.spectrograms is not None:\n            spec = self.spectrograms.get(samplename, None)\n        else:\n            spec = None\n\n        if spec is None:\n            spec = np.zeros(self.cfg.TARGET_SHAPE, dtype=np.float32)\n\n        if isinstance(spec, np.memmap):\n            spec = np.array(spec, dtype=np.float32)\n            \n        spec = torch.tensor(spec, dtype=torch.float32).unsqueeze(0)  # Add channel dimension\n\n        if self.mode == \"train\" and random.random() < self.cfg.aug_prob:\n            spec = self.apply_spec_augmentations(spec)\n        \n        target = self.encode_label(row['primary_label'])\n        \n        if 'secondary_labels' in row and row['secondary_labels'] not in [[''], None, np.nan]:\n            if isinstance(row['secondary_labels'], str):\n                secondary_labels = eval(row['secondary_labels'])\n            else:\n                secondary_labels = row['secondary_labels']\n            \n            for label in secondary_labels:\n                if label in self.label_to_idx:\n                    target[self.label_to_idx[label]] = 1.0\n        \n        return {\n            'melspec': spec, \n            'target': torch.tensor(target, dtype=torch.float32),\n            'filename': row['filename']\n        }\n    \n    def apply_spec_augmentations(self, spec):\n        \"\"\"Apply enhanced augmentations to spectrogram\"\"\"\n        # Time masking (horizontal stripes)\n        if random.random() < 0.7:\n            num_masks = random.randint(1, 5)  # Increased from 3\n            for _ in range(num_masks):\n                width = random.randint(5, 30)  # Increased max width from 20\n                start = random.randint(0, spec.shape[2] - width)\n                spec[:, :, start:start+width] = 0\n        \n        # Frequency masking (vertical stripes)\n        if random.random() < 0.7:\n            num_masks = random.randint(1, 5)  # Increased from 3\n            for _ in range(num_masks):\n                height = random.randint(5, 30)  # Increased max height from 20\n                start = random.randint(0, spec.shape[1] - height)\n                spec[:, start:start+height, :] = 0\n        \n        # Random brightness/contrast\n        if random.random() < 0.7:\n            gain = random.uniform(0.8, 1.2)\n            bias = random.uniform(-0.1, 0.1)\n            spec = spec * gain + bias\n            spec = torch.clamp(spec, 0, 1)\n            \n        # Gaussian noise\n        if random.random() < 0.5:\n            noise = torch.randn_like(spec) * random.uniform(0.001, 0.02)\n            spec = spec + noise\n            spec = torch.clamp(spec, 0, 1)\n            \n        # Time shifting (roll horizontally)\n        if random.random() < 0.5:\n            shift = random.randint(-spec.shape[2]//8, spec.shape[2]//8)\n            if shift != 0:\n                spec = torch.roll(spec, shifts=shift, dims=2)\n                \n        # Frequency shifting (roll vertically)\n        if random.random() < 0.3:\n            shift = random.randint(-spec.shape[1]//20, spec.shape[1]//20)\n            if shift != 0:\n                spec = torch.roll(spec, shifts=shift, dims=1)\n                \n        return spec\n    \n    def encode_label(self, label):\n        \"\"\"Encode label to one-hot vector\"\"\"\n        target = np.zeros(self.num_classes)\n        if label in self.label_to_idx:\n            target[self.label_to_idx[label]] = 1.0\n        return target\n\ndef collate_fn(batch):\n    \"\"\"Custom collate function to handle different sized spectrograms\"\"\"\n    batch = [item for item in batch if item is not None]\n    if len(batch) == 0:\n        return {}\n        \n    result = {key: [] for key in batch[0].keys()}\n    \n    for item in batch:\n        for key, value in item.items():\n            result[key].append(value)\n    \n    for key in result:\n        if key == 'target' and isinstance(result[key][0], torch.Tensor):\n            result[key] = torch.stack(result[key])\n        elif key == 'melspec' and isinstance(result[key][0], torch.Tensor):\n            shapes = [t.shape for t in result[key]]\n            if len(set(str(s) for s in shapes)) == 1:\n                result[key] = torch.stack(result[key])\n    \n    return result\n\nclass BirdCLEFModel(nn.Module):\n    def __init__(self, cfg):\n        super().__init__()\n        self.cfg = cfg\n        \n        taxonomy_df = pd.read_csv(cfg.taxonomy_csv)\n        cfg.num_classes = len(taxonomy_df)\n        \n        # For Kaggle: create model with or without pretrained weights\n        print(f\"Creating model: {cfg.model_name}\")\n        try:\n            self.backbone = timm.create_model(\n                cfg.model_name,\n                pretrained=cfg.pretrained,\n                in_chans=cfg.in_channels,\n                drop_rate=0.2,  # Lower dropout for MobileNetV3\n                drop_path_rate=0.2  # Lower stochastic depth for MobileNetV3\n            )\n            print(f\"Successfully created {cfg.model_name}\")\n            # Print available methods and attributes for debugging\n            print(f\"Model structure: {type(self.backbone)}\")\n            if hasattr(self.backbone, 'classifier'):\n                print(f\"Classifier: {self.backbone.classifier}\")\n        except Exception as e:\n            print(f\"Error creating model: {e}\")\n            # Try alternative model name formats\n            alternative_names = [\n                'mobilenetv3_small.100_in1k',  # Alternative name in newer timm\n                'tf_mobilenetv3_small_100',    # TF variant\n                'mobilenetv3_small'            # Simplified name\n            ]\n            for alt_name in alternative_names:\n                try:\n                    print(f\"Trying alternative model name: {alt_name}\")\n                    self.backbone = timm.create_model(\n                        alt_name,\n                        pretrained=False,\n                        in_chans=cfg.in_channels\n                    )\n                    # Update config to match successful model\n                    cfg.model_name = alt_name\n                    print(f\"Successfully created {alt_name}\")\n                    break\n                except Exception as e2:\n                    print(f\"Error with {alt_name}: {e2}\")\n        \n        # Load pretrained weights from local file if specified\n        if not cfg.pretrained and cfg.pretrained_weights:\n            print(f\"Loading pretrained weights from: {cfg.pretrained_weights}\")\n            try:\n                state_dict = torch.load(cfg.pretrained_weights, map_location='cpu')\n                # Handle case where state_dict might contain 'model' or 'state_dict' key\n                if 'model' in state_dict:\n                    state_dict = state_dict['model']\n                elif 'state_dict' in state_dict:\n                    state_dict = state_dict['state_dict']\n                \n                # Remove prefix if it exists (like 'backbone.')\n                if all(k.startswith('backbone.') for k in state_dict if k not in ['cls_token', 'pos_embed']):\n                    state_dict = {k.replace('backbone.', ''): v for k, v in state_dict.items()}\n                \n                # Remove classifier weights\n                for k in list(state_dict.keys()):\n                    if 'classifier' in k or 'fc' in k or 'head' in k:\n                        del state_dict[k]\n                \n                self.backbone.load_state_dict(state_dict, strict=False)\n                print(\"Successfully loaded pretrained weights\")\n            except Exception as e:\n                print(f\"Error loading pretrained weights: {e}\")\n        \n        # Debug available classifier structures\n        print(f\"Available attributes: {dir(self.backbone)}\")\n        \n        try:\n            if 'efficientnet' in cfg.model_name:\n                backbone_out = self.backbone.classifier.in_features\n                self.backbone.classifier = nn.Identity()\n                print(f\"Using EfficientNet classifier with {backbone_out} features\")\n            elif 'resnet' in cfg.model_name:\n                backbone_out = self.backbone.fc.in_features\n                self.backbone.fc = nn.Identity()\n                print(f\"Using ResNet classifier with {backbone_out} features\")\n            elif 'mobilenetv3' in cfg.model_name:\n                # MobileNetV3 classifier structure can vary between timm versions\n                if hasattr(self.backbone, 'classifier') and hasattr(self.backbone.classifier, 'in_features'):\n                    backbone_out = self.backbone.classifier.in_features\n                    self.backbone.classifier = nn.Identity()\n                    print(f\"Using MobileNetV3 standard classifier with {backbone_out} features\")\n                elif hasattr(self.backbone, 'classifier') and isinstance(self.backbone.classifier, nn.Sequential):\n                    # For MobileNetV3 with sequential classifier\n                    backbone_out = 0  # Initialize before loop\n                    for module in self.backbone.classifier:\n                        if isinstance(module, nn.Linear):\n                            backbone_out = module.in_features\n                            break\n                    if backbone_out == 0:\n                        backbone_out = 1280  # Default for MobileNetV3 small\n                    self.backbone.classifier = nn.Identity()\n                    print(f\"Using MobileNetV3 sequential classifier with {backbone_out} features\")\n                elif hasattr(self.backbone, 'head') and hasattr(self.backbone.head, 'fc'):\n                    backbone_out = self.backbone.head.fc.in_features\n                    self.backbone.head.fc = nn.Identity()\n                    print(f\"Using MobileNetV3 head.fc with {backbone_out} features\")\n                else:\n                    # Fallback to typical mobilenetv3 small dimension\n                    backbone_out = 1280  # Standard size for MobileNetV3 Small\n                    if hasattr(self.backbone, 'classifier'):\n                        self.backbone.classifier = nn.Identity()\n                    print(f\"Using fallback MobileNetV3 feature dimension: {backbone_out}\")\n            else:\n                # Try to get classifier info for other models\n                print(\"Using generic classifier detection\")\n                if hasattr(self.backbone, 'get_classifier') and callable(getattr(self.backbone, 'get_classifier')):\n                    backbone_out = self.backbone.get_classifier().in_features\n                    self.backbone.reset_classifier(0, '')\n                else:\n                    # Last resort - find any linear layer as a hint\n                    backbone_out = 0\n                    for name, module in self.backbone.named_modules():\n                        if isinstance(module, nn.Linear):\n                            backbone_out = module.in_features\n                            print(f\"Found linear layer with {backbone_out} features: {name}\")\n                            # Don't break, we want the last one\n                    \n                    if backbone_out == 0:\n                        backbone_out = 1280  # Default fallback\n                    print(f\"Using fallback feature dimension: {backbone_out}\")\n        except Exception as e:\n            print(f\"Error setting up classifier: {e}\")\n            # Fallback to a reasonable size for MobileNetV3\n            backbone_out = 1280\n            print(f\"Using emergency fallback feature dimension: {backbone_out}\")\n        \n        self.pooling = nn.AdaptiveAvgPool2d(1)\n        \n        # Add attention mechanism\n        self.attention = nn.Sequential(\n            nn.Conv2d(backbone_out, backbone_out // 16, kernel_size=1),\n            nn.SiLU(),\n            nn.Conv2d(backbone_out // 16, backbone_out, kernel_size=1),\n            nn.Sigmoid()\n        )\n            \n        self.feat_dim = backbone_out\n        \n        # Add multi-sample dropout for better generalization\n        self.dropouts = nn.ModuleList([\n            nn.Dropout(0.3) for _ in range(5)\n        ])\n        \n        self.classifier = nn.Linear(backbone_out, cfg.num_classes)\n        \n        self.mixup_enabled = hasattr(cfg, 'mixup_alpha') and cfg.mixup_alpha > 0\n        self.cutmix_enabled = hasattr(cfg, 'cutmix_alpha') and cfg.cutmix_alpha > 0\n        \n        if self.mixup_enabled:\n            self.mixup_alpha = cfg.mixup_alpha\n        if self.cutmix_enabled:\n            self.cutmix_alpha = cfg.cutmix_alpha\n            \n    def forward(self, x, targets=None):\n        b = x.size(0)\n        \n        # Apply mixup or cutmix during training\n        if self.training and targets is not None:\n            if self.mixup_enabled and self.cutmix_enabled:\n                # Randomly choose between mixup and cutmix\n                if random.random() < 0.5:\n                    x, targets_a, targets_b, lam = self.mixup_data(x, targets)\n                else:\n                    x, targets_a, targets_b, lam = self.cutmix_data(x, targets)\n            elif self.mixup_enabled:\n                x, targets_a, targets_b, lam = self.mixup_data(x, targets)\n            elif self.cutmix_enabled:\n                x, targets_a, targets_b, lam = self.cutmix_data(x, targets)\n            else:\n                targets_a, targets_b, lam = targets, targets, 1.0\n        else:\n            targets_a, targets_b, lam = None, None, None\n        \n        features = self.backbone(x)\n        \n        # Handle different output formats from different backbones\n        if isinstance(features, dict):\n            features = features['features']\n        \n        # For MobileNetV3 and other models, ensure we have 4D tensor for attention\n        # If features is already flattened (2D), reshape it to 4D for attention\n        if len(features.shape) == 2:\n            # Create pseudo spatial dimensions\n            features = features.unsqueeze(-1).unsqueeze(-1)\n            \n        # Now features should be 4D, apply attention mechanism\n        att = self.attention(features)\n        features = features * att\n        \n        # Pool and flatten\n        features = self.pooling(features)\n        features = features.view(b, -1)\n        \n        # Multi-sample dropout for robust training\n        if self.training:\n            logits = torch.zeros(b, self.cfg.num_classes).to(features.device)\n            for dropout in self.dropouts:\n                logits += self.classifier(dropout(features))\n            logits /= len(self.dropouts)\n        else:\n            logits = self.classifier(features)\n        \n        if self.training and (self.mixup_enabled or self.cutmix_enabled) and targets is not None:\n            loss = self.mixup_criterion(F.binary_cross_entropy_with_logits, \n                                       logits, targets_a, targets_b, lam)\n            return logits, loss\n            \n        return logits\n    \n    def mixup_data(self, x, targets):\n        \"\"\"Applies mixup to the data batch\"\"\"\n        batch_size = x.size(0)\n\n        lam = np.random.beta(self.mixup_alpha, self.mixup_alpha)\n\n        indices = torch.randperm(batch_size).to(x.device)\n\n        mixed_x = lam * x + (1 - lam) * x[indices]\n        \n        return mixed_x, targets, targets[indices], lam\n    \n    def cutmix_data(self, x, targets):\n        \"\"\"Applies cutmix to the data batch\"\"\"\n        batch_size = x.size(0)\n        lam = np.random.beta(self.cutmix_alpha, self.cutmix_alpha)\n        \n        # Generate random box\n        W, H = x.size(2), x.size(3)\n        cut_ratio = np.sqrt(1.0 - lam)\n        cut_w = int(W * cut_ratio)\n        cut_h = int(H * cut_ratio)\n        \n        # Uniform\n        cx = np.random.randint(W)\n        cy = np.random.randint(H)\n        \n        # Limit box to image\n        bby1 = np.clip(cy - cut_h // 2, 0, H)\n        bbx1 = np.clip(cx - cut_w // 2, 0, W)\n        bby2 = np.clip(cy + cut_h // 2, 0, H)\n        bbx2 = np.clip(cx + cut_w // 2, 0, W)\n        \n        # Random sample\n        rand_index = torch.randperm(batch_size).to(x.device)\n        \n        # Apply cutmix - first verify the indices are valid\n        x_cut = x.clone()\n        \n        # Only apply if the box has valid dimensions\n        if bbx2 > bbx1 and bby2 > bby1:\n            x_cut[:, :, bbx1:bbx2, bby1:bby2] = x[rand_index, :, bbx1:bbx2, bby1:bby2]\n            \n            # Adjust lambda\n            cut_area = (bbx2 - bbx1) * (bby2 - bby1)\n            lam = 1.0 - (cut_area / (W * H))\n        else:\n            print(f\"Warning: Invalid cutmix box dimensions ({bbx1},{bby1})-({bbx2},{bby2})\")\n        \n        return x_cut, targets, targets[rand_index], lam\n    \n    def mixup_criterion(self, criterion, pred, y_a, y_b, lam):\n        \"\"\"Applies mixup to the loss function\"\"\"\n        return lam * criterion(pred, y_a) + (1 - lam) * criterion(pred, y_b)\n\ndef get_optimizer(model, cfg):\n  \n    if cfg.optimizer == 'Adam':\n        optimizer = optim.Adam(\n            model.parameters(),\n            lr=cfg.lr,\n            weight_decay=cfg.weight_decay\n        )\n    elif cfg.optimizer == 'AdamW':\n        optimizer = optim.AdamW(\n            model.parameters(),\n            lr=cfg.lr,\n            weight_decay=cfg.weight_decay\n        )\n    elif cfg.optimizer == 'SGD':\n        optimizer = optim.SGD(\n            model.parameters(),\n            lr=cfg.lr,\n            momentum=0.9,\n            weight_decay=cfg.weight_decay\n        )\n    else:\n        raise NotImplementedError(f\"Optimizer {cfg.optimizer} not implemented\")\n        \n    return optimizer\n\ndef get_scheduler(optimizer, cfg):\n   \n    if cfg.scheduler == 'CosineAnnealingLR':\n        scheduler = lr_scheduler.CosineAnnealingLR(\n            optimizer,\n            T_max=cfg.T_max,\n            eta_min=cfg.min_lr\n        )\n    elif cfg.scheduler == 'CosineAnnealingWarmRestarts':\n        scheduler = lr_scheduler.CosineAnnealingWarmRestarts(\n            optimizer,\n            T_0=cfg.T_0,\n            T_mult=cfg.T_mult,\n            eta_min=cfg.min_lr\n        )\n    elif cfg.scheduler == 'ReduceLROnPlateau':\n        scheduler = lr_scheduler.ReduceLROnPlateau(\n            optimizer,\n            mode='min',\n            factor=0.5,\n            patience=2,\n            min_lr=cfg.min_lr,\n            verbose=True\n        )\n    elif cfg.scheduler == 'StepLR':\n        scheduler = lr_scheduler.StepLR(\n            optimizer,\n            step_size=cfg.epochs // 3,\n            gamma=0.5\n        )\n    elif cfg.scheduler == 'OneCycleLR':\n        scheduler = None  \n    else:\n        scheduler = None\n        \n    return scheduler\n\ndef get_criterion(cfg):\n \n    if cfg.criterion == 'BCEWithLogitsLoss':\n        criterion = nn.BCEWithLogitsLoss()\n    else:\n        raise NotImplementedError(f\"Criterion {cfg.criterion} not implemented\")\n        \n    return criterion\n\ndef train_one_epoch(model, loader, optimizer, criterion, device, scheduler=None, scaler=None):\n    \n    model.train()\n    losses = []\n    all_targets = []\n    all_outputs = []\n    optimizer.zero_grad()\n    \n    pbar = tqdm(enumerate(loader), total=len(loader), desc=\"Training\")\n    \n    for step, batch in pbar:\n        # Skip empty batches\n        if not batch:\n            continue\n            \n        if isinstance(batch['melspec'], list):\n            batch_outputs = []\n            batch_losses = []\n            \n            for i in range(len(batch['melspec'])):\n                inputs = batch['melspec'][i].unsqueeze(0).to(device)\n                target = batch['target'][i].unsqueeze(0).to(device)\n                \n                if cfg.USE_AMP:\n                    with autocast():\n                        output = model(inputs)\n                        loss = criterion(output, target)\n                    \n                    scaler.scale(loss).backward()\n                    batch_outputs.append(output.detach().cpu())\n                    batch_losses.append(loss.item())\n                else:\n                    output = model(inputs)\n                    loss = criterion(output, target)\n                    loss.backward()\n                    batch_outputs.append(output.detach().cpu())\n                    batch_losses.append(loss.item())\n            \n            if (step + 1) % cfg.gradient_accumulation_steps == 0:\n                if cfg.USE_AMP:\n                    scaler.step(optimizer)\n                    scaler.update()\n                else:\n                    optimizer.step()\n                optimizer.zero_grad()\n                \n            outputs = torch.cat(batch_outputs, dim=0).numpy()\n            loss = np.mean(batch_losses)\n            targets = batch['target'].numpy()\n            \n        else:\n            inputs = batch['melspec'].to(device)\n            targets = batch['target'].to(device)\n            \n            if cfg.USE_AMP:\n                with autocast():\n                    outputs = model(inputs)\n                    \n                    if isinstance(outputs, tuple):\n                        outputs, loss = outputs  \n                    else:\n                        loss = criterion(outputs, targets)\n                \n                scaler.scale(loss).backward()\n                \n                if (step + 1) % cfg.gradient_accumulation_steps == 0:\n                    scaler.step(optimizer)\n                    scaler.update()\n                    optimizer.zero_grad()\n            else:\n                outputs = model(inputs)\n                \n                if isinstance(outputs, tuple):\n                    outputs, loss = outputs  \n                else:\n                    loss = criterion(outputs, targets)\n                    \n                loss.backward()\n                \n                if (step + 1) % cfg.gradient_accumulation_steps == 0:\n                    optimizer.step()\n                    optimizer.zero_grad()\n            \n            outputs = outputs.detach().cpu().numpy()\n            targets = targets.detach().cpu().numpy()\n        \n        if scheduler is not None and isinstance(scheduler, lr_scheduler.OneCycleLR):\n            scheduler.step()\n            \n        all_outputs.append(outputs)\n        all_targets.append(targets)\n        losses.append(loss if isinstance(loss, float) else loss.item())\n        \n        pbar.set_postfix({\n            'train_loss': np.mean(losses[-10:]) if losses else 0,\n            'lr': optimizer.param_groups[0]['lr']\n        })\n    \n    all_outputs = np.concatenate(all_outputs)\n    all_targets = np.concatenate(all_targets)\n    auc = calculate_auc(all_targets, all_outputs)\n    avg_loss = np.mean(losses)\n    \n    return avg_loss, auc\n\n@torch.no_grad()\ndef validate(model, loader, criterion, device):\n   \n    model.eval()\n    losses = []\n    all_targets = []\n    all_outputs = []\n    \n    for batch in tqdm(loader, desc=\"Validation\"):\n        # Skip empty batches\n        if not batch:\n            continue\n            \n        if isinstance(batch['melspec'], list):\n            batch_outputs = []\n            batch_losses = []\n            \n            for i in range(len(batch['melspec'])):\n                inputs = batch['melspec'][i].unsqueeze(0).to(device)\n                target = batch['target'][i].unsqueeze(0).to(device)\n                \n                if cfg.USE_AMP:\n                    with autocast():\n                        output = model(inputs)\n                        loss = criterion(output, target)\n                else:\n                    output = model(inputs)\n                    loss = criterion(output, target)\n                \n                batch_outputs.append(output.detach().cpu())\n                batch_losses.append(loss.item())\n            \n            outputs = torch.cat(batch_outputs, dim=0).numpy()\n            loss = np.mean(batch_losses)\n            targets = batch['target'].numpy()\n                \n        else:\n            inputs = batch['melspec'].to(device)\n            targets = batch['target'].to(device)\n            \n            if cfg.USE_AMP:\n                with autocast():\n                    outputs = model(inputs)\n                    loss = criterion(outputs, targets)\n            else:\n                outputs = model(inputs)\n                loss = criterion(outputs, targets)\n            \n            outputs = outputs.detach().cpu().numpy()\n            targets = targets.detach().cpu().numpy()\n        \n        all_outputs.append(outputs)\n        all_targets.append(targets)\n        losses.append(loss if isinstance(loss, float) else loss.item())\n    \n    all_outputs = np.concatenate(all_outputs)\n    all_targets = np.concatenate(all_targets)\n    \n    auc = calculate_auc(all_targets, all_outputs)\n    avg_loss = np.mean(losses)\n    \n    return avg_loss, auc\n\ndef calculate_auc(targets, outputs):\n    \"\"\"Optimized AUC calculation\"\"\"\n    num_classes = targets.shape[1]\n    probs = 1 / (1 + np.exp(-outputs))\n    \n    # Vectorized approach for classes with positive samples\n    aucs = []\n    active_classes = np.where(np.sum(targets, axis=0) > 0)[0]\n    \n    for i in active_classes:\n        class_auc = roc_auc_score(targets[:, i], probs[:, i])\n        aucs.append(class_auc)\n    \n    return np.mean(aucs) if aucs else 0.0\n\ndef run_training(df, cfg):\n    \"\"\"Training function that can either use pre-computed spectrograms or generate them on-the-fly\"\"\"\n\n    # Ensure torch is imported in the local scope\n    import torch\n    import torch.nn as nn\n    from torch.cuda.amp import autocast, GradScaler\n    from torch.optim import lr_scheduler\n\n    taxonomy_df = pd.read_csv(cfg.taxonomy_csv)\n    species_ids = taxonomy_df['primary_label'].tolist()\n    cfg.num_classes = len(species_ids)\n    \n    if cfg.debug:\n        cfg.update_debug_settings()\n\n    # Handle spectrograms loading\n    spectrograms = None\n    if cfg.spectrogram_npy and os.path.exists(cfg.spectrogram_npy):\n        print(f\"Loading pre-computed mel spectrograms from NPY file: {cfg.spectrogram_npy}\")\n        try:\n            # Use improved loading function\n            spectrograms = load_spectrograms_mmap(cfg.spectrogram_npy)\n            print(f\"Successfully loaded {len(spectrograms)} pre-computed spectrograms\")\n        except Exception as e:\n            print(f\"Error loading pre-computed spectrograms: {e}\")\n            spectrograms = None\n    else:\n        print(\"No pre-computed spectrograms available. Will generate on-the-fly.\")\n        # You could add code here to generate spectrograms if needed\n    \n    if 'filepath' not in df.columns:\n        df['filepath'] = cfg.train_datadir + '/' + df.filename\n    if 'samplename' not in df.columns:\n        df['samplename'] = df.filename.map(lambda x: x.split('/')[0] + '-' + x.split('/')[-1].split('.')[0])\n        \n    skf = StratifiedKFold(n_splits=cfg.n_fold, shuffle=True, random_state=cfg.seed)\n    \n    best_scores = []\n    \n    for fold, (train_idx, val_idx) in enumerate(skf.split(df, df['primary_label'])):\n        if fold not in cfg.selected_folds:\n            continue\n            \n        print(f'\\n{\"=\"*30} Fold {fold} {\"=\"*30}')\n        \n        train_df = df.iloc[train_idx].reset_index(drop=True)\n        val_df = df.iloc[val_idx].reset_index(drop=True)\n        \n        print(f'Training set: {len(train_df)} samples')\n        print(f'Validation set: {len(val_df)} samples')\n        \n        train_dataset = BirdCLEFDatasetFromNPY(train_df, cfg, spectrograms=spectrograms, mode='train')\n        val_dataset = BirdCLEFDatasetFromNPY(val_df, cfg, spectrograms=spectrograms, mode='valid')\n        \n        train_loader = DataLoader(\n            train_dataset, \n            batch_size=cfg.batch_size, \n            shuffle=True, \n            num_workers=cfg.num_workers,\n            pin_memory=cfg.PIN_MEMORY,\n            collate_fn=collate_fn,\n            drop_last=True,\n            persistent_workers=cfg.num_workers > 0\n        )\n        \n        val_loader = DataLoader(\n            val_dataset, \n            batch_size=cfg.batch_size, \n            shuffle=False, \n            num_workers=cfg.num_workers,\n            pin_memory=cfg.PIN_MEMORY,\n            collate_fn=collate_fn,\n            persistent_workers=cfg.num_workers > 0\n        )\n        \n        model = BirdCLEFModel(cfg).to(cfg.device)\n        \n        # Try to enable torch.compile for PyTorch 2.0+\n        if hasattr(torch, 'compile') and torch.__version__ >= '2.0.0':\n            try:\n                # Set dynamo config to suppress errors and fall back to eager mode\n                import torch._dynamo\n                torch._dynamo.config.suppress_errors = True\n                \n                # You can set USE_COMPILE = False to disable compilation completely\n                USE_COMPILE = True\n                \n                if USE_COMPILE:\n                    model = torch.compile(model, backend='eager')  # Use 'eager' backend instead of default 'inductor'\n                    print(\"Using torch.compile() with eager backend for JIT acceleration\")\n            except Exception as e:\n                print(f\"torch.compile() failed: {e}, using standard model\")\n                \n        optimizer = get_optimizer(model, cfg)\n        criterion = get_criterion(cfg)\n        \n        if cfg.scheduler == 'OneCycleLR':\n            scheduler = lr_scheduler.OneCycleLR(\n                optimizer,\n                max_lr=cfg.lr,\n                steps_per_epoch=len(train_loader) // cfg.gradient_accumulation_steps,\n                epochs=cfg.epochs,\n                pct_start=0.1\n            )\n        else:\n            scheduler = get_scheduler(optimizer, cfg)\n        \n        # Initialize AMP scaler\n        scaler = GradScaler() if cfg.USE_AMP else None\n        \n        best_auc = 0\n        best_epoch = 0\n        \n        for epoch in range(cfg.epochs):\n            print(f\"\\nEpoch {epoch+1}/{cfg.epochs}\")\n            \n            train_loss, train_auc = train_one_epoch(\n                model, \n                train_loader, \n                optimizer, \n                criterion, \n                cfg.device,\n                scheduler if isinstance(scheduler, lr_scheduler.OneCycleLR) else None,\n                scaler\n            )\n            \n            val_loss, val_auc = validate(model, val_loader, criterion, cfg.device)\n\n            if scheduler is not None and not isinstance(scheduler, lr_scheduler.OneCycleLR):\n                if isinstance(scheduler, lr_scheduler.ReduceLROnPlateau):\n                    scheduler.step(val_loss)\n                else:\n                    scheduler.step()\n\n            print(f\"Train Loss: {train_loss:.4f}, Train AUC: {train_auc:.4f}\")\n            print(f\"Val Loss: {val_loss:.4f}, Val AUC: {val_auc:.4f}\")\n            \n            if val_auc > best_auc:\n                best_auc = val_auc\n                best_epoch = epoch + 1\n                print(f\"New best AUC: {best_auc:.4f} at epoch {best_epoch}\")\n\n                torch.save({\n                    'model_state_dict': model.state_dict(),\n                    'optimizer_state_dict': optimizer.state_dict(),\n                    'scheduler_state_dict': scheduler.state_dict() if scheduler else None,\n                    'epoch': epoch,\n                    'val_auc': val_auc,\n                    'train_auc': train_auc,\n                    'cfg': cfg\n                }, f\"model_fold{fold}.pth\")\n        \n        best_scores.append(best_auc)\n        print(f\"\\nBest AUC for fold {fold}: {best_auc:.4f} at epoch {best_epoch}\")\n        \n        # Clear memory\n        del model, optimizer, scheduler, train_loader, val_loader\n        torch.cuda.empty_cache()\n        gc.collect()\n    \n    print(\"\\n\" + \"=\"*60)\n    print(\"Cross-Validation Results:\")\n    for fold, score in enumerate(best_scores):\n        print(f\"Fold {cfg.selected_folds[fold]}: {score:.4f}\")\n    print(f\"Mean AUC: {np.mean(best_scores):.4f}\")\n    print(\"=\"*60)\n\ndef save_model(model, optimizer=None, scheduler=None, epoch=0, val_auc=0, train_auc=0, cfg=None, path=\"./bird_model.pth\"):\n    \"\"\"\n    Save model, optimizer, scheduler states and configuration to the specified path\n    \"\"\"\n    # Get model state dict and handle compiled model case\n    state_dict = model.state_dict()\n    \n    # Handle compiled model state dict (keys with \"_orig_mod.\" prefix)\n    fixed_state_dict = {}\n    for k, v in state_dict.items():\n        if k.startswith('_orig_mod.'):\n            fixed_state_dict[k[10:]] = v  # Remove '_orig_mod.' prefix (10 characters)\n        else:\n            fixed_state_dict[k] = v\n    \n    save_dict = {\n        'model_state_dict': fixed_state_dict,\n        'epoch': epoch,\n        'val_auc': val_auc,\n        'train_auc': train_auc\n    }\n    \n    if optimizer is not None:\n        save_dict['optimizer_state_dict'] = optimizer.state_dict()\n    \n    if scheduler is not None:\n        save_dict['scheduler_state_dict'] = scheduler.state_dict()\n        \n    if cfg is not None:\n        save_dict['cfg'] = cfg\n        \n    torch.save(save_dict, path)\n    print(f\"Model saved to {path}\")\n\nif __name__ == \"__main__\":\n    print(\"Starting BirdCLEF training with MobileNetV3...\")\n    \n    # Check if timm is properly installed and can access models\n    try:\n        import timm\n        print(f\"TIMM version: {timm.__version__}\")\n        \n        # List available MobileNetV3 models\n        available_models = [m for m in timm.list_models() if 'mobilenetv3' in m.lower()]\n        print(f\"Available MobileNetV3 models in timm: {available_models}\")\n        \n        if not available_models:\n            print(\"No MobileNetV3 models found, installing latest timm version...\")\n            import subprocess\n            subprocess.run([\"pip\", \"install\", \"-U\", \"timm\"], check=True)\n            print(\"Timm updated. Please restart the notebook/script.\")\n    except Exception as e:\n        print(f\"Error with timm: {e}\")\n        print(\"Installing timm...\")\n        import subprocess\n        subprocess.run([\"pip\", \"install\", \"timm\"], check=True)\n        print(\"Please restart the notebook/script after timm installation.\")\n    \n    # Try to load and process the data\n    try:\n        if os.path.exists(cfg.train_csv):\n            print(f\"Loading train data from {cfg.train_csv}\")\n            df = pd.read_csv(cfg.train_csv)\n            print(f\"Loaded {len(df)} training samples\")\n            \n            # Run training\n            run_training(df, cfg)\n        else:\n            print(f\"Training CSV not found at {cfg.train_csv}\")\n            # Use relative paths for local testing if Kaggle paths not available\n            if not os.path.exists('./train.csv'):\n                print(\"No training data found. Please make sure the data is available.\")\n            else:\n                print(\"Using local data paths...\")\n                cfg.train_datadir = './train_audio'  \n                cfg.train_csv = './train.csv'\n                cfg.taxonomy_csv = './taxonomy.csv'\n                df = pd.read_csv(cfg.train_csv)\n                run_training(df, cfg)\n    except Exception as e:\n        print(f\"Error running training: {e}\")\n        import traceback\n        traceback.print_exc()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-20T13:54:37.300045Z","iopub.execute_input":"2025-05-20T13:54:37.300414Z","iopub.status.idle":"2025-05-20T13:57:20.826832Z","shell.execute_reply.started":"2025-05-20T13:54:37.300386Z","shell.execute_reply":"2025-05-20T13:57:20.825442Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}