{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"isSourceIdPinned":false,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"##### IMPORTS #####\n\"\"\"\nPyTorch-based neural networks for bird call classification\n\"\"\"\nimport os\nimport logging\nimport random\nimport time\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport librosa\nimport joblib\nimport cv2\nfrom pathlib import Path\nimport glob\n\n# Core ML imports\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import (confusion_matrix, classification_report, accuracy_score, \n                           precision_recall_fscore_support)\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\n\n# PyTorch imports\ntry:\n    import torch\n    import torch.nn as nn\n    import torch.nn.functional as F\n    import torch.optim as optim\n    from torch.utils.data import Dataset, DataLoader, TensorDataset\n    from torch.optim.lr_scheduler import StepLR, ReduceLROnPlateau\n    import torch.backends.cudnn as cudnn\n    PYTORCH_AVAILABLE = True\n    print(\"✓ PyTorch loaded successfully\")\n    \n    # GPU Optimization Configuration\n    print(\"🔧 Configuring GPU optimizations...\")\n    \n    # Check GPU availability\n    if torch.cuda.is_available():\n        device = torch.device('cuda')\n        print(f\"✓ GPU Configuration Complete:\")\n        print(f\"   • GPU Device: {torch.cuda.get_device_name()}\")\n        print(f\"   • CUDA Version: {torch.version.cuda}\")\n        print(f\"   • GPU Memory: {torch.cuda.get_device_properties(0).total_memory / 1e9:.1f} GB\")\n        \n        # Enable cuDNN optimizations\n        cudnn.benchmark = True\n        cudnn.deterministic = False\n        print(\"✓ cuDNN optimizations enabled\")\n    else:\n        device = torch.device('cpu')\n        print(\"⚠️ No GPU detected, using CPU\")\n    \n    print(f\"✓ Device set to: {device}\")\n    \nexcept ImportError:\n    PYTORCH_AVAILABLE = False\n    device = 'cpu'\n    print(\"⚠️ PyTorch not available. Neural networks will be skipped.\")\n\nfrom tqdm.auto import tqdm\n\n# Performance optimizations\nwarnings.filterwarnings('ignore')\nwarnings.filterwarnings('ignore', category=UserWarning, module='sklearn')\nplt.style.use('default')\n\n##### CONFIGURATION #####\n\"\"\"\nEnhanced configuration for neural network architectures and deep learning\n\"\"\"\nclass CFG:\n    # Seed for reproducibility\n    seed = 42\n    debug = True\n    \n    # Data paths (Kaggle paths)\n    train_datadir = '/kaggle/input/birdclef-2025/train_audio'\n    train_csv = '/kaggle/input/birdclef-2025/train.csv'\n    taxonomy_csv = '/kaggle/input/birdclef-2025/taxonomy.csv'\n    \n    # Audio processing parameters (optimized for speed)\n    FS = 16000  # Reduced from 32000 for faster processing\n    TARGET_DURATION = 5.0  \n    \n    # Mel spectrogram parameters (improved dimensions)\n    N_FFT = 1024  # Increased for better frequency resolution\n    HOP_LENGTH = 512  # Increased for better time resolution\n    N_MELS = 128  # More mel bands for better feature representation\n    FMIN = 50\n    FMAX = 8000  # Adjusted based on FS/2\n    TARGET_SHAPE = (128, 128)  # Larger dimensions for better features\n    \n    # Data Augmentation Settings (easy toggle)\n    enable_data_augmentation = True\n    augmentation_probability = 0.3  # 30% chance to apply each augmentation\n    \n    # Augmentation parameters\n    noise_factor = 0.02  # Background noise intensity\n    volume_range = (0.7, 1.3)  # Volume scaling range\n    mixup_alpha = 0.2  # Mixup parameter\n    \n    # Quality filtering\n    filter_low_quality = True  # Remove samples with rating 0.5-2.5\n    min_quality_rating = 2.5  # Minimum rating to keep\n    \n    # Performance parameters\n    n_samples = 1500 if debug else None  # Small sample for quick testing\n    min_samples_for_rare_class_elimination = 10  # Higher threshold\n    test_size = 0.2\n    \n    # Neural Network Training Parameters\n    batch_size = 32  # Reduced batch size for better stability\n    epochs = 50  # Reduced epochs for faster testing\n    learning_rate = 0.0001  # Reduced learning rate for better convergence\n    dropout_rate = 0.2  # Reduced dropout rate\n    early_stopping_patience = 8  # Reduced patience\n    \n    # Neural Network Architectures (Simplified for better performance)\n    neural_architectures = {\n        'CNN_Simple': {\n            'type': 'cnn',\n            'conv_channels': [32, 64], \n            'kernel_sizes': [3, 3],    \n            'pool_sizes': [2, 2],      \n            'fc_layers': [128],        # Reduced FC layer size\n            'dropout_rate': 0.2\n        },\n        'CNN_Medium': {\n            'type': 'cnn',\n            'conv_channels': [32, 64, 128],\n            'kernel_sizes': [3, 3, 3],\n            'pool_sizes': [2, 2, 2],\n            'fc_layers': [256],  # Reduced FC layer size\n            'dropout_rate': 0.25\n        },\n        'LSTM_Fixed': {\n            'type': 'lstm',\n            'hidden_sizes': [64], # Simplified LSTM\n            'num_layers': 1,  # Reduced layers\n            'bidirectional': False,  # Simplified to unidirectional\n            'fc_layers': [128], # Reduced FC layer size\n            'dropout_rate': 0.2,\n            'lstm_seq_len': 128, # Changed to make input divisible (16384/128=128)\n        }\n    }\n    \n    # Ensemble Configuration\n    enable_ensemble = True\n    ensemble_method = 'voting'  # 'voting' or 'weighted'\n    ensemble_weights = None  # Will be calculated based on validation performance\n    \n    # GPU and Performance Optimization Parameters\n    use_multiprocessing = True  # Enable parallel processing\n    n_jobs = -1  # Use all available cores\n    enable_mixed_precision = True  # Use mixed precision training\n    clear_memory_between_models = True  # Clear memory between model trainings\n    reduce_model_architectures = debug  # Use fewer architectures in debug mode\n\n##### UTILITY FUNCTIONS #####\n\"\"\"\nHelper functions and setup\n\"\"\"\ndef set_seed(seed=42):\n    \"\"\"Set seed for reproducibility\"\"\"\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    if PYTORCH_AVAILABLE:\n        torch.manual_seed(seed)\n        torch.cuda.manual_seed(seed)\n        torch.cuda.manual_seed_all(seed)\n    print(f\"✓ Seed set: {seed}\")\n\ndef setup_logging():\n    \"\"\"Logging setup\"\"\"\n    logging.basicConfig(\n        level=logging.INFO,\n        format='%(asctime)s - %(levelname)s - %(message)s',\n        handlers=[\n            logging.StreamHandler(),\n            logging.FileHandler('training_log.log')\n        ]\n    )\n    print(\"=\"*60)\n    print(\"🚀 BirdCLEF Neural Network Training Pipeline Started\")\n    print(\"=\"*60)\n\ndef print_section(title):\n    \"\"\"Helper function for printing section titles\"\"\"\n    print(\"\\n\" + \"=\"*60)\n    print(f\"📊 {title}\")\n    print(\"=\"*60)\n\ndef print_configuration(cfg):\n    \"\"\"Print configuration settings\"\"\"\n    print_section(\"⚙️ CONFIGURATION USED\")\n    \n    config_text = f\"\"\"\n⚙️ Configuration Used:\n• Data Augmentation: {'✓ Enabled' if cfg.enable_data_augmentation else '✗ Disabled'}\n• Quality Filtering: {'✓ Enabled' if cfg.filter_low_quality else '✗ Disabled'}\n\n📊 Data Parameters:\n• Sample Rate: {cfg.FS} Hz\n• Target Duration: {cfg.TARGET_DURATION}s\n• N_FFT: {cfg.N_FFT}\n• Hop Length: {cfg.HOP_LENGTH}\n• N_Mels: {cfg.N_MELS}\n• Target Shape: {cfg.TARGET_SHAPE}\n\n🧠 Neural Network Parameters:\n• Batch Size: {cfg.batch_size}\n• Epochs: {cfg.epochs}\n• Learning Rate: {cfg.learning_rate}\n• Dropout Rate: {cfg.dropout_rate}\n• Early Stopping Patience: {cfg.early_stopping_patience}\n• Test Size: {cfg.test_size*100}%\n• Debug Mode: {'✓ Enabled' if cfg.debug else '✗ Disabled'}\n• Sample Limit: {cfg.n_samples if cfg.debug else 'None (All data)'}\n\n⚡ GPU & Performance Optimizations:\n• Device: {device if PYTORCH_AVAILABLE else 'CPU (PyTorch not available)'}\n• Multiprocessing: {'✓ Enabled' if getattr(cfg, 'use_multiprocessing', False) else '✗ Disabled'}\n• Mixed Precision: {'✓ Enabled' if getattr(cfg, 'enable_mixed_precision', False) else '✗ Disabled'}\n• Memory Cleanup: {'✓ Enabled' if getattr(cfg, 'clear_memory_between_models', False) else '✗ Disabled'}\n\n🏗️ Neural Network Architectures: {len(cfg.neural_architectures)}\n\"\"\"\n    \n    for arch_name, arch_config in cfg.neural_architectures.items():\n        config_text += f\"• {arch_name} ({arch_config['type'].upper()}): \"\n        if arch_config['type'] == 'cnn':\n            config_text += f\"layers={arch_config['conv_channels']}, kernel_sizes={arch_config['kernel_sizes']}, pool_sizes={arch_config['pool_sizes']}\\n\"\n        elif arch_config['type'] == 'lstm':\n            config_text += f\"hidden={arch_config['hidden_sizes']}, layers={arch_config['num_layers']}\\n\"\n    \n    config_text += f\"\"\"\n🎯 Ensemble Configuration:\n• Ensemble Enabled: {'✓ Yes' if cfg.enable_ensemble else '✗ No'}\n• Ensemble Method: {cfg.ensemble_method}\n    \"\"\"\n    \n    print(config_text)\n\n##### DATA AUGMENTATION FUNCTIONS #####\n\"\"\"\nAudio data augmentation techniques\n\"\"\"\ndef add_background_noise(audio, noise_factor=0.02):\n    \"\"\"Add Gaussian background noise\"\"\"\n    noise = np.random.normal(0, noise_factor, len(audio))\n    return audio + noise\n\ndef volume_scaling(audio, volume_range=(0.7, 1.3)):\n    \"\"\"Apply random volume scaling\"\"\"\n    factor = np.random.uniform(volume_range[0], volume_range[1])\n    return audio * factor\n\ndef mixup_audio(audio1, audio2, alpha=0.2):\n    \"\"\"Apply mixup augmentation between two audio samples\"\"\"\n    lam = np.random.beta(alpha, alpha)\n    \n    # Ensure both audio samples have the same length\n    min_len = min(len(audio1), len(audio2))\n    audio1 = audio1[:min_len]\n    audio2 = audio2[:min_len]\n    \n    mixed_audio = lam * audio1 + (1 - lam) * audio2\n    return mixed_audio, lam\n\ndef apply_augmentation(audio, cfg, aug_type=None):\n    \"\"\"Apply random augmentation to audio\"\"\"\n    if not cfg.enable_data_augmentation:\n        return audio\n    \n    # Apply augmentation with probability\n    if np.random.random() > cfg.augmentation_probability:\n        return audio\n    \n    augmented_audio = audio.copy()\n    \n    # Background noise\n    if aug_type is None or aug_type == 'noise':\n        if np.random.random() < 0.5:\n            augmented_audio = add_background_noise(augmented_audio, cfg.noise_factor)\n    \n    # Volume scaling\n    if aug_type is None or aug_type == 'volume':\n        if np.random.random() < 0.5:\n            augmented_audio = volume_scaling(augmented_audio, cfg.volume_range)\n    \n    return augmented_audio\n\n##### ENHANCED AUDIO PROCESSING #####\n\"\"\"\nEnhanced audio processing functions with improved feature extraction\n\"\"\"\ndef extract_enhanced_audio_features(audio_path, cfg, apply_augmentation_flag=True):\n    \"\"\"\n    Enhanced audio feature extraction with multiple feature types:\n    - Mel spectrogram (primary features)\n    - MFCC features (cepstral coefficients)\n    - Spectral features (centroid, rolloff, zero crossing rate)\n    - Rhythm features (tempo, beat density)\n    \"\"\"\n    try:\n        y, sr = librosa.load(audio_path, sr=cfg.FS, duration=cfg.TARGET_DURATION)\n        \n        if len(y) == 0:\n            # Calculate correct feature dimension - FIXED to be divisible by 128\n            base_size = cfg.TARGET_SHAPE[0] * cfg.TARGET_SHAPE[1]  # 128*128 = 16384\n            return np.zeros(base_size, dtype=np.float32)  # Simplified to only mel features\n        \n        # Normalize audio\n        y = librosa.util.normalize(y)\n        \n        # Apply data augmentation if enabled\n        if apply_augmentation_flag and cfg.enable_data_augmentation:\n            y = apply_augmentation(y, cfg)\n        \n        # 1. Mel spectrogram (primary features) - ONLY use mel features for now\n        melspec = librosa.feature.melspectrogram(\n            y=y, sr=sr, n_fft=cfg.N_FFT, hop_length=cfg.HOP_LENGTH,\n            n_mels=cfg.N_MELS, fmin=cfg.FMIN, fmax=cfg.FMAX\n        )\n        melspec = librosa.power_to_db(melspec, ref=np.max)\n        melspec = (melspec - melspec.min()) / (melspec.max() - melspec.min() + 1e-8)\n        melspec = cv2.resize(melspec, (cfg.TARGET_SHAPE[1], cfg.TARGET_SHAPE[0]), \n                           interpolation=cv2.INTER_AREA)\n        \n        # Return only mel spectrogram features for now (16384 features)\n        melspec_features = melspec.flatten()\n        return melspec_features.astype(np.float32)\n        \n    except Exception as e:\n        print(f\"⚠️ Audio processing error for {audio_path}: {e}\")\n        # Return zeros with correct dimension - 16384 features\n        base_size = cfg.TARGET_SHAPE[0] * cfg.TARGET_SHAPE[1]\n        return np.zeros(base_size, dtype=np.float32)\n\n##### NEURAL NETWORK ARCHITECTURES #####\n\"\"\"\nPyTorch neural network architectures for bird call classification\n\"\"\"\n\nclass SimpleCNN(nn.Module):\n    \"\"\"Simplified CNN architecture for audio classification\"\"\"\n    def __init__(self, input_dim, num_classes, cfg, arch_config):\n        super(SimpleCNN, self).__init__()\n        \n        self.input_dim = input_dim\n        self.num_classes = num_classes\n        self.cfg = cfg \n        self.arch_config = arch_config \n        \n        # Melspectrogram dimensions\n        self.melspec_h = cfg.TARGET_SHAPE[0]\n        self.melspec_w = cfg.TARGET_SHAPE[1]\n        self.melspec_features = self.melspec_h * self.melspec_w\n\n        # Simplified: Only handle mel spectrogram features (no additional features)\n        assert input_dim == self.melspec_features, f\"Expected input_dim {self.melspec_features}, got {input_dim}\"\n\n        # Convolutional layers\n        conv_layers = []\n        in_channels = 1 # Start with 1 channel for the melspectrogram image\n        \n        current_h, current_w = self.melspec_h, self.melspec_w\n\n        for i, out_channels in enumerate(self.arch_config['conv_channels']):\n            kernel_size = self.arch_config['kernel_sizes'][i]\n            pool_size = self.arch_config['pool_sizes'][i]\n            \n            conv_layers.append(nn.Conv2d(in_channels, out_channels, kernel_size=kernel_size, padding=kernel_size//2))\n            conv_layers.append(nn.BatchNorm2d(out_channels))\n            conv_layers.append(nn.ReLU())\n            conv_layers.append(nn.MaxPool2d(kernel_size=pool_size, stride=pool_size))\n            in_channels = out_channels\n            \n            # Update dimensions\n            current_h = current_h // pool_size\n            current_w = current_w // pool_size\n            \n        self.conv_block = nn.Sequential(*conv_layers)\n        \n        # Calculate the size after conv layers\n        self.conv_output_features = in_channels * current_h * current_w\n        \n        # Fully connected layers - simplified\n        fc_layers_list = []\n        fc_input_dim = self.conv_output_features\n        \n        for fc_hidden_size in self.arch_config['fc_layers']:\n            fc_layers_list.append(nn.Linear(fc_input_dim, fc_hidden_size))\n            fc_layers_list.append(nn.ReLU())\n            fc_layers_list.append(nn.Dropout(self.arch_config['dropout_rate']))\n            fc_input_dim = fc_hidden_size\n            \n        fc_layers_list.append(nn.Linear(fc_input_dim, num_classes))\n        self.fc_block = nn.Sequential(*fc_layers_list)\n        \n    def forward(self, x):\n        batch_size = x.size(0)\n        \n        # Reshape melspectrogram for CNN (only mel features now)\n        melspec_data = x.view(batch_size, 1, self.melspec_h, self.melspec_w)\n        \n        # Pass through conv block\n        conv_out = self.conv_block(melspec_data)\n        \n        # Flatten conv output\n        conv_out_flat = conv_out.view(batch_size, -1)\n        \n        # Pass through FC block\n        output = self.fc_block(conv_out_flat)\n        \n        return output\n\nclass BirdLSTM(nn.Module):\n    \"\"\"Enhanced LSTM architecture for sequential audio features\"\"\"\n    def __init__(self, input_dim, num_classes, cfg, arch_config):\n        super(BirdLSTM, self).__init__()\n        \n        self.input_dim = input_dim\n        self.num_classes = num_classes\n        self.cfg = cfg\n        self.arch_config = arch_config\n        \n        self.lstm_seq_len = arch_config.get('lstm_seq_len', 64) # Default if not in config\n        \n        # Ensure features_per_step is an integer and input_dim is divisible\n        if input_dim % self.lstm_seq_len != 0:\n            raise ValueError(f\"input_dim ({input_dim}) must be divisible by lstm_seq_len ({self.lstm_seq_len})\")\n        self.features_per_step = input_dim // self.lstm_seq_len\n\n        # LSTM layers\n        self.lstm = nn.LSTM(\n            input_size=self.features_per_step, # Corrected input size\n            hidden_size=arch_config['hidden_sizes'][0], # Use first hidden_size, or handle list for multi-layer\n            num_layers=arch_config['num_layers'],\n            dropout=arch_config['dropout_rate'] if arch_config['num_layers'] > 1 else 0,\n            bidirectional=arch_config['bidirectional'],\n            batch_first=True\n        )\n        \n        # Calculate LSTM output size\n        lstm_output_size = arch_config['hidden_sizes'][0] * 2 if arch_config['bidirectional'] else arch_config['hidden_sizes'][0]\n        \n        # Fully connected layers\n        fc_layers_list = []\n        fc_input_dim = lstm_output_size\n        \n        for fc_hidden_size in arch_config['fc_layers']:\n            fc_layers_list.append(nn.Linear(fc_input_dim, fc_hidden_size))\n            fc_layers_list.append(nn.ReLU())\n            fc_layers_list.append(nn.Dropout(arch_config['dropout_rate']))\n            fc_input_dim = fc_hidden_size\n            \n        fc_layers_list.append(nn.Linear(fc_input_dim, num_classes))\n        self.fc_block = nn.Sequential(*fc_layers_list)\n        \n    def forward(self, x):\n        batch_size = x.size(0)\n        \n        # Reshape for LSTM: (batch, seq_len, features_per_step)\n        # The input x is expected to be flat (batch_size, input_dim)\n        x = x.view(batch_size, self.lstm_seq_len, self.features_per_step)\n        \n        # LSTM forward pass\n        lstm_out, (hidden, cell) = self.lstm(x)\n        \n        # Use the hidden state of the last time step\n        if self.arch_config['bidirectional']:\n            # Concatenate final forward and backward hidden states of the last layer\n            # hidden is (num_layers * num_directions, batch, hidden_size)\n            # We want the hidden state from the last layer, so indices -2 (forward) and -1 (backward)\n            hidden_last_layer_fwd = hidden[-2, :, :] \n            hidden_last_layer_bwd = hidden[-1, :, :]\n            processed_lstm_out = torch.cat((hidden_last_layer_fwd, hidden_last_layer_bwd), dim=1)\n        else:\n            # hidden is (num_layers, batch, hidden_size)\n            # We want the hidden state from the last layer, so index -1\n            processed_lstm_out = hidden[-1, :, :]\n        \n        # Pass through FC block\n        output = self.fc_block(processed_lstm_out)\n        \n        return output\n\n##### ENSEMBLE METHODS #####\n\"\"\"\nEnsemble methods for combining multiple models\n\"\"\"\n\nclass EnsembleModel:\n    \"\"\"Ensemble of multiple neural network models\"\"\"\n    def __init__(self, models, method='voting', weights=None):\n        self.models = models\n        self.method = method\n        self.weights = weights\n        \n    def predict(self, X):\n        \"\"\"Make predictions using ensemble of models\"\"\"\n        predictions = []\n        \n        for model in self.models:\n            model.eval()\n            with torch.no_grad():\n                if isinstance(X, np.ndarray):\n                    X_tensor = torch.FloatTensor(X).to(device)\n                else:\n                    X_tensor = X.to(device)\n                    \n                output = model(X_tensor)\n                if self.method == 'voting':\n                    pred = output.argmax(dim=1).cpu().numpy()\n                else:\n                    pred = F.softmax(output, dim=1).cpu().numpy()\n                predictions.append(pred)\n        \n        predictions = np.array(predictions)\n        \n        if self.method == 'voting':\n            # Hard voting\n            ensemble_pred = []\n            for i in range(predictions.shape[1]):\n                votes = predictions[:, i]\n                ensemble_pred.append(np.bincount(votes).argmax())\n            return np.array(ensemble_pred)\n        \n        elif self.method == 'weighted':\n            # Weighted average\n            if self.weights is None:\n                self.weights = np.ones(len(self.models)) / len(self.models)\n            \n            weighted_probs = np.average(predictions, axis=0, weights=self.weights)\n            return weighted_probs.argmax(axis=1)\n        \n    def predict_proba(self, X):\n        \"\"\"Get probability predictions from ensemble\"\"\"\n        predictions = []\n        \n        for model in self.models:\n            model.eval()\n            with torch.no_grad():\n                if isinstance(X, np.ndarray):\n                    X_tensor = torch.FloatTensor(X).to(device)\n                else:\n                    X_tensor = X.to(device)\n                    \n                output = model(X_tensor)\n                probs = F.softmax(output, dim=1).cpu().numpy()\n                predictions.append(probs)\n        \n        predictions = np.array(predictions)\n        \n        if self.weights is None:\n            return np.mean(predictions, axis=0)\n        else:\n            return np.average(predictions, axis=0, weights=self.weights)\n\n##### DATA PREPARATION #####\n\"\"\"\nData preparation and preprocessing\n\"\"\"\ndef prepare_data(cfg):\n    \"\"\"Enhanced data preparation for neural networks\"\"\"\n    print_section(\"DATA PREPARATION\")\n    \n    # Create dummy data for testing if files don't exist\n    if not os.path.exists(cfg.train_csv):\n        print(\"⚠️ Data files not found. Creating demo data...\")\n        return create_dummy_data(cfg)\n    \n    # Load data\n    print(\"📂 Loading data files...\")\n    train_df = pd.read_csv(cfg.train_csv)\n    \n    # Load taxonomy data if available\n    if os.path.exists(cfg.taxonomy_csv):\n        taxonomy_df = pd.read_csv(cfg.taxonomy_csv)\n        print(f\"✓ Loaded taxonomy data: {len(taxonomy_df)} species\")\n        \n        # Merge with train data to get additional species information\n        train_df = train_df.merge(taxonomy_df, left_on='primary_label', right_on='primary_label', how='left')\n    \n    print(f\"✓ {len(train_df)} samples loaded\")\n    \n    # Debug mode sampling\n    if cfg.debug and cfg.n_samples and cfg.n_samples < len(train_df):\n        train_df = train_df.sample(cfg.n_samples, random_state=cfg.seed)\n        print(f\"🔬 Debug mode: {cfg.n_samples} samples selected\")\n    \n    # Enhanced data filtering\n    print(\"🧹 Cleaning data...\")\n    \n    # Remove samples with missing primary labels\n    initial_len = len(train_df)\n    train_df = train_df.dropna(subset=['primary_label'])\n    if len(train_df) < initial_len:\n        print(f\"✓ Removed {initial_len - len(train_df)} samples with missing labels\")\n    \n    # Filter by quality rating (remove low quality samples)\n    if 'rating' in train_df.columns and cfg.filter_low_quality:\n        initial_len = len(train_df)\n        # Remove samples with rating between 0.5 and 2.5\n        quality_filtered = train_df[~((train_df['rating'] >= 0.5) & (train_df['rating'] <= cfg.min_quality_rating))]\n        train_df = quality_filtered\n        print(f\"✓ Filtered low quality samples (rating 0.5-{cfg.min_quality_rating}): removed {initial_len - len(train_df)} samples\")\n    \n    # Remove rare classes\n    class_counts = train_df['primary_label'].value_counts()\n    rare_classes = class_counts[class_counts < cfg.min_samples_for_rare_class_elimination].index\n    \n    if len(rare_classes) > 0:\n        initial_len = len(train_df)\n        train_df = train_df[~train_df['primary_label'].isin(rare_classes)]\n        print(f\"✓ {len(rare_classes)} rare classes eliminated ({initial_len - len(train_df)} samples)\")\n    \n    # Create file paths\n    train_df['filepath'] = train_df['filename'].apply(lambda x: os.path.join(cfg.train_datadir, x))\n    \n    # Encode labels\n    le = LabelEncoder()\n    train_df['target'] = le.fit_transform(train_df['primary_label'])\n    \n    # Save label encoder\n    joblib.dump(le, \"label_encoder.joblib\")\n    \n    # Update config\n    cfg.num_classes = len(le.classes_)\n    cfg.class_names = le.classes_\n    \n    print(f\"✓ {cfg.num_classes} classes, {len(train_df)} samples ready\")\n    \n    # Enhanced class distribution analysis\n    class_dist = train_df['primary_label'].value_counts()\n    print(f\"📊 Top 10 most common classes: {dict(class_dist.head(10))}\")\n    print(f\"📊 Class distribution stats: min={class_dist.min()}, max={class_dist.max()}, mean={class_dist.mean():.1f}\")\n    \n    return train_df, le\n\ndef create_dummy_data(cfg):\n    \"\"\"Create dummy data for demo purposes\"\"\"\n    print(\"🎭 Creating demo data...\")\n    \n    # Create dummy audio files and dataframe\n    n_samples = 500\n    bird_species = ['robin', 'sparrow', 'eagle', 'hawk', 'crow', 'owl', 'cardinal', 'bluejay', 'woodpecker', 'finch']\n    \n    data = []\n    for i in range(n_samples):\n        species = np.random.choice(bird_species)\n        filename = f\"{species}_{i:03d}.wav\"\n        # Add some metadata with realistic rating distribution\n        rating = np.random.choice([0, 1, 2, 3, 4, 5], p=[0.1, 0.1, 0.2, 0.3, 0.2, 0.1])\n        data.append({\n            'filename': filename,\n            'primary_label': species,\n            'rating': rating,\n            'latitude': np.random.uniform(-90, 90),\n            'longitude': np.random.uniform(-180, 180),\n            'author': f\"user_{np.random.randint(1, 50)}\"\n        })\n    \n    train_df = pd.DataFrame(data)\n    train_df['filepath'] = train_df['filename']  # Use dummy paths\n    \n    # Encode labels\n    le = LabelEncoder()\n    train_df['target'] = le.fit_transform(train_df['primary_label'])\n    \n    # Save label encoder\n    joblib.dump(le, \"label_encoder.joblib\")\n    \n    # Update config\n    cfg.num_classes = len(le.classes_)\n    cfg.class_names = le.classes_\n    \n    print(f\"✓ Demo data ready: {cfg.num_classes} classes, {len(train_df)} samples\")\n    \n    return train_df, le\n\ndef extract_features(df, cfg):\n    \"\"\"Enhanced feature extraction with progress tracking and augmentation\"\"\"\n    print_section(\"FEATURE EXTRACTION\")\n    \n    print(\"🎵 Extracting enhanced audio features...\")\n    \n    # Display data augmentation configuration\n    if cfg.enable_data_augmentation:\n        print(\"\\n🔄 Data Augmentation Applied:\")\n        print(f\"   • Background Noise: ✓ Enabled (factor: {cfg.noise_factor})\")\n        print(f\"   • Volume Scaling: ✓ Enabled (range: {cfg.volume_range})\")\n        print(f\"   • Mixup Alpha: {cfg.mixup_alpha}\")\n        print(f\"   • Augmentation Probability: {cfg.augmentation_probability * 100}%\")\n        print(\"   • Multi-modal Features: Mel + MFCC + Spectral + Rhythm\")\n    else:\n        print(\"\\n🚫 Data Augmentation: Disabled\")\n    \n    # For demo data, create more sophisticated random features\n    if not os.path.exists(cfg.train_datadir):\n        print(\"🎭 Creating enhanced demo features...\")\n        \n        # Calculate feature dimensions - FIXED\n        base_size = cfg.TARGET_SHAPE[0] * cfg.TARGET_SHAPE[1]  # 128*128 = 16384\n        n_features = base_size  # Only mel features\n        \n        print(f\"🚀 Using optimized vectorized feature generation... Features: {n_features}\")\n        \n        # Vectorized feature generation for speed\n        n_samples = len(df)\n        y = df['target'].values\n        \n        # Pre-allocate arrays for better memory usage\n        X = np.zeros((n_samples, n_features), dtype=np.float32)\n        \n        # Create more realistic patterns for each species\n        n_classes = len(np.unique(y))\n        \n        # Create species-specific templates for better separation\n        species_templates = np.random.randn(n_classes, n_features).astype(np.float32) * 0.5\n        \n        # Generate more realistic features\n        for i, target in enumerate(y):\n            # Base pattern from species template\n            base_pattern = species_templates[target]\n            \n            # Add noise for variation\n            noise = np.random.randn(n_features).astype(np.float32) * 0.2\n            \n            # Create frequency patterns (simulate mel spectrogram structure)\n            mel_h, mel_w = cfg.TARGET_SHAPE\n            pattern_2d = base_pattern.reshape(mel_h, mel_w)\n            \n            # Add frequency-specific patterns (lower frequencies more active for bird calls)\n            freq_weights = np.linspace(1.0, 0.3, mel_h).reshape(-1, 1)\n            pattern_2d = pattern_2d * freq_weights\n            \n            # Add temporal patterns\n            time_weights = np.ones((1, mel_w))\n            time_weights[0, mel_w//4:3*mel_w//4] *= 1.5  # More activity in middle time frames\n            pattern_2d = pattern_2d * time_weights\n            \n            # Flatten and add noise\n            final_pattern = pattern_2d.flatten() + noise\n            \n            # Normalize to reasonable range\n            final_pattern = np.tanh(final_pattern)  # Bound to [-1, 1]\n            \n            X[i] = final_pattern\n        \n        print(f\"✓ Enhanced demo feature matrix: {X.shape} (realistic patterns)\")\n        return X, y\n    \n    # Real feature extraction\n    features = []\n    labels = []\n    \n    start_time = time.time()\n    \n    # Sequential processing\n    print(\"🔄 Using sequential feature extraction...\")\n    for idx, row in tqdm(df.iterrows(), total=len(df), desc=\"Processing Audio\"):\n        try:\n            feature_vector = extract_enhanced_audio_features(row['filepath'], cfg)\n            \n            if not np.all(feature_vector == 0):\n                features.append(feature_vector)\n                labels.append(row['target'])\n                \n        except Exception as e:\n            print(f\"⚠️ Skipping {row['filepath']}: {e}\")\n            continue\n    \n    X = np.array(features, dtype=np.float32)\n    y = np.array(labels)\n    \n    elapsed_time = time.time() - start_time\n    print(f\"✓ Feature extraction completed: {X.shape}, {elapsed_time:.2f} seconds\")\n    print(f\"⚡ Processing speed: {len(X)/elapsed_time:.2f} samples/second\")\n    \n    if X.shape[0] == 0:\n        raise ValueError(\"❌ No samples remaining!\")\n    \n    return X, y\n\n##### NEURAL NETWORK TRAINING #####\n\"\"\"\nPyTorch neural network training functions\n\"\"\"\n\ndef create_model(arch_name, arch_config, input_dim, num_classes, cfg):\n    \"\"\"Create a neural network model based on architecture configuration\"\"\"\n    if not PYTORCH_AVAILABLE:\n        return None\n    \n    if arch_config['type'] == 'cnn':\n        model = SimpleCNN(input_dim, num_classes, cfg, arch_config)\n    elif arch_config['type'] == 'lstm':\n        model = BirdLSTM(\n            input_dim, num_classes, cfg, arch_config\n        )\n    else:\n        raise ValueError(f\"Unknown architecture type: {arch_config['type']}\")\n    \n    return model.to(device)\n\ndef train_model(model, train_loader, val_loader, cfg, model_name):\n    \"\"\"Train a single neural network model\"\"\"\n    print(f\"🚀 Training {model_name}...\")\n    \n    # Loss function and optimizer\n    criterion = nn.CrossEntropyLoss()\n    optimizer = optim.Adam(model.parameters(), lr=cfg.learning_rate)\n    scheduler = ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=5, verbose=True)\n    \n    # Training history\n    train_losses = []\n    val_losses = []\n    train_accuracies = []\n    val_accuracies = []\n    \n    best_val_loss = float('inf')\n    patience_counter = 0\n    \n    start_time = time.time()\n    \n    for epoch in range(cfg.epochs):\n        # Training phase\n        model.train()\n        train_loss = 0.0\n        train_correct = 0\n        train_total = 0\n        \n        for batch_idx, (data, target) in enumerate(train_loader):\n            data, target = data.to(device), target.to(device)\n            \n            optimizer.zero_grad()\n            \n            # Mixed precision training if enabled and CUDA available\n            if cfg.enable_mixed_precision and torch.cuda.is_available():\n                with torch.cuda.amp.autocast():\n                    output = model(data)\n                    loss = criterion(output, target)\n                \n                scaler = torch.cuda.amp.GradScaler()\n                scaler.scale(loss).backward()\n                scaler.step(optimizer)\n                scaler.update()\n            else:\n                output = model(data)\n                loss = criterion(output, target)\n                loss.backward()\n                optimizer.step()\n            \n            train_loss += loss.item()\n            _, predicted = output.max(1)\n            train_total += target.size(0)\n            train_correct += predicted.eq(target).sum().item()\n        \n        # Validation phase\n        model.eval()\n        val_loss = 0.0\n        val_correct = 0\n        val_total = 0\n        \n        with torch.no_grad():\n            for data, target in val_loader:\n                data, target = data.to(device), target.to(device)\n                output = model(data)\n                loss = criterion(output, target)\n                \n                val_loss += loss.item()\n                _, predicted = output.max(1)\n                val_total += target.size(0)\n                val_correct += predicted.eq(target).sum().item()\n        \n        # Calculate metrics\n        avg_train_loss = train_loss / len(train_loader)\n        avg_val_loss = val_loss / len(val_loader)\n        train_acc = 100. * train_correct / train_total\n        val_acc = 100. * val_correct / val_total\n        \n        train_losses.append(avg_train_loss)\n        val_losses.append(avg_val_loss)\n        train_accuracies.append(train_acc)\n        val_accuracies.append(val_acc)\n        \n        # Learning rate scheduling\n        scheduler.step(avg_val_loss)\n        \n        # Early stopping\n        if avg_val_loss < best_val_loss:\n            best_val_loss = avg_val_loss\n            patience_counter = 0\n            # Save best model\n            torch.save(model.state_dict(), f'best_{model_name.lower()}.pth')\n        else:\n            patience_counter += 1\n        \n        # Print progress\n        if (epoch + 1) % 10 == 0 or epoch == 0:\n            print(f\"   Epoch {epoch+1:3d}/{cfg.epochs}: \"\n                  f\"Train Loss: {avg_train_loss:.4f}, Train Acc: {train_acc:.2f}%, \"\n                  f\"Val Loss: {avg_val_loss:.4f}, Val Acc: {val_acc:.2f}%\")\n        \n        # Early stopping check\n        if patience_counter >= cfg.early_stopping_patience:\n            print(f\"   Early stopping at epoch {epoch+1}\")\n            break\n    \n    training_time = time.time() - start_time\n    \n    # Load best model\n    model.load_state_dict(torch.load(f'best_{model_name.lower()}.pth'))\n    \n    return {\n        'model': model,\n        'train_losses': train_losses,\n        'val_losses': val_losses,\n        'train_accuracies': train_accuracies,\n        'val_accuracies': val_accuracies,\n        'training_time': training_time,\n        'epochs_trained': len(train_losses)\n    }\n\ndef evaluate_model(model, test_loader, label_encoder, model_name):\n    \"\"\"Evaluate a trained model\"\"\"\n    model.eval()\n    test_correct = 0\n    test_total = 0\n    all_predictions = []\n    all_targets = []\n    \n    with torch.no_grad():\n        for data, target in test_loader:\n            data, target = data.to(device), target.to(device)\n            output = model(data)\n            _, predicted = output.max(1)\n            \n            test_total += target.size(0)\n            test_correct += predicted.eq(target).sum().item()\n            \n            all_predictions.extend(predicted.cpu().numpy())\n            all_targets.extend(target.cpu().numpy())\n    \n    accuracy = 100. * test_correct / test_total\n    \n    # Calculate detailed metrics\n    precision, recall, f1, _ = precision_recall_fscore_support(\n        all_targets, all_predictions, average='weighted', zero_division=0\n    )\n    \n    print(f\"📊 {model_name} Test Results:\")\n    print(f\"   Accuracy: {accuracy:.2f}%\")\n    print(f\"   Precision: {precision:.4f}\")\n    print(f\"   Recall: {recall:.4f}\")\n    print(f\"   F1-Score: {f1:.4f}\")\n    \n    return {\n        'accuracy': accuracy,\n        'precision': precision,\n        'recall': recall,\n        'f1': f1,\n        'predictions': all_predictions,\n        'targets': all_targets\n    }\n\ndef train_neural_networks(X_train, y_train, X_test, y_test, cfg, label_encoder):\n    \"\"\"Train all neural network architectures\"\"\"\n    if not PYTORCH_AVAILABLE:\n        print(\"⚠️ PyTorch not available. Skipping neural network training.\")\n        return {}\n    \n    print_section(\"NEURAL NETWORK TRAINING\")\n    \n    # Prepare data\n    print(\"📊 Preparing data for neural networks...\")\n    \n    # Scale features\n    scaler = StandardScaler()\n    X_train_scaled = scaler.fit_transform(X_train).astype(np.float32)\n    X_test_scaled = scaler.transform(X_test).astype(np.float32)\n    \n    # Convert to PyTorch tensors\n    X_train_tensor = torch.FloatTensor(X_train_scaled)\n    y_train_tensor = torch.LongTensor(y_train)\n    X_test_tensor = torch.FloatTensor(X_test_scaled)\n    y_test_tensor = torch.LongTensor(y_test)\n    \n    # Create data loaders\n    train_dataset = TensorDataset(X_train_tensor, y_train_tensor)\n    test_dataset = TensorDataset(X_test_tensor, y_test_tensor)\n    \n    # Split training data for validation\n    train_size = int(0.8 * len(train_dataset))\n    val_size = len(train_dataset) - train_size\n    train_subset, val_subset = torch.utils.data.random_split(train_dataset, [train_size, val_size])\n    \n    train_loader = DataLoader(train_subset, batch_size=cfg.batch_size, shuffle=True)\n    val_loader = DataLoader(val_subset, batch_size=cfg.batch_size, shuffle=False)\n    test_loader = DataLoader(test_dataset, batch_size=cfg.batch_size, shuffle=False)\n    \n    print(f\"✓ Data loaders ready: Train: {len(train_loader)}, Val: {len(val_loader)}, Test: {len(test_loader)}\")\n    \n    # Train models\n    models = {}\n    results = {}\n    \n    input_dim = X_train.shape[1]\n    num_classes = cfg.num_classes\n    \n    for arch_name, arch_config in cfg.neural_architectures.items():\n        print(f\"\\n🔄 Training {arch_name}...\")\n        \n        try:\n            # Create model\n            model = create_model(arch_name, arch_config, input_dim, num_classes, cfg)\n            \n            if model is None:\n                continue\n            \n            print(f\"   Model parameters: {sum(p.numel() for p in model.parameters()):,}\")\n            \n            # Train model\n            training_result = train_model(model, train_loader, val_loader, cfg, arch_name)\n            \n            # Evaluate model\n            eval_result = evaluate_model(model, test_loader, label_encoder, arch_name)\n            \n            # Combine results\n            results[arch_name] = {\n                **training_result,\n                **eval_result,\n                'architecture': arch_config\n            }\n            \n            models[arch_name] = model\n            \n            # Memory cleanup\n            if cfg.clear_memory_between_models:\n                torch.cuda.empty_cache()\n            \n        except Exception as e:\n            print(f\"   ❌ Error training {arch_name}: {e}\")\n            continue\n    \n    # Save models\n    for arch_name, model in models.items():\n        model_path = f\"{arch_name.lower()}_model.pth\"\n        torch.save(model.state_dict(), model_path)\n        print(f\"✓ Saved {arch_name} model to {model_path}\")\n    \n    return models, results\n\ndef create_ensemble(models, cfg, X_test, y_test):\n    \"\"\"Create and evaluate ensemble model\"\"\"\n    if not models or not cfg.enable_ensemble:\n        return None\n    \n    print_section(\"ENSEMBLE MODEL\")\n    \n    model_list = list(models.values())\n    ensemble = EnsembleModel(model_list, method=cfg.ensemble_method)\n    \n    # Evaluate ensemble\n    print(\"🔄 Evaluating ensemble model...\")\n    \n    # Scale test data\n    scaler = StandardScaler()\n    scaler.fit(X_test)  # Note: In practice, use the same scaler from training\n    X_test_scaled = scaler.transform(X_test)\n    \n    ensemble_predictions = ensemble.predict(X_test_scaled)\n    ensemble_accuracy = accuracy_score(y_test, ensemble_predictions)\n    \n    precision, recall, f1, _ = precision_recall_fscore_support(\n        y_test, ensemble_predictions, average='weighted', zero_division=0\n    )\n    \n    print(f\"🎯 Ensemble Results:\")\n    print(f\"   Method: {cfg.ensemble_method}\")\n    print(f\"   Models: {len(model_list)}\")\n    print(f\"   Accuracy: {ensemble_accuracy:.4f}\")\n    print(f\"   Precision: {precision:.4f}\")\n    print(f\"   Recall: {recall:.4f}\")\n    print(f\"   F1-Score: {f1:.4f}\")\n    \n    # Save ensemble\n    ensemble_path = f\"ensemble_{cfg.ensemble_method}.pkl\"\n    joblib.dump(ensemble, ensemble_path)\n    print(f\"✓ Saved ensemble model to {ensemble_path}\")\n    \n    return ensemble, {\n        'accuracy': ensemble_accuracy,\n        'precision': precision,\n        'recall': recall,\n        'f1': f1,\n        'predictions': ensemble_predictions\n    }\n\n##### MAIN PIPELINE #####\n\"\"\"\nMain pipeline execution\n\"\"\"\ndef main():\n    \"\"\"Enhanced main pipeline for neural network training\"\"\"\n    setup_logging()\n    set_seed(CFG.seed)\n    \n    print_section(\"🚀 BirdCLEF Neural Network Pipeline Starting\")\n    \n    # Display configuration\n    print_configuration(CFG)\n    \n    total_start_time = time.time()\n    \n    try:\n        # 1. Data Preparation\n        train_df, label_encoder = prepare_data(CFG)\n        \n        # 2. Feature Extraction\n        X, y = extract_features(train_df, CFG)\n        \n        # 3. Train-Test Split\n        print_section(\"DATA SPLITTING\")\n        print(\"📊 Splitting into training and test sets...\")\n        \n        # Check if stratification is possible\n        unique, counts = np.unique(y, return_counts=True)\n        min_class_count = counts.min()\n        \n        if min_class_count >= 2:\n            X_train, X_test, y_train, y_test = train_test_split(\n                X, y, test_size=CFG.test_size, random_state=CFG.seed, \n                stratify=y\n            )\n            print(\"✓ Stratified split used\")\n        else:\n            X_train, X_test, y_train, y_test = train_test_split(\n                X, y, test_size=CFG.test_size, random_state=CFG.seed\n            )\n            print(\"⚠️ Stratified split not possible (insufficient samples)\")\n        \n        print(f\"✓ Training: {X_train.shape}, Test: {X_test.shape}\")\n        \n        # 4. Neural Network Training\n        models, results = train_neural_networks(X_train, y_train, X_test, y_test, CFG, label_encoder)\n        \n        if not models:\n            print(\"❌ No models were successfully trained!\")\n            return\n        \n        # 5. Ensemble Creation\n        ensemble, ensemble_results = create_ensemble(models, CFG, X_test, y_test)\n        \n        # 6. Results Summary\n        total_time = time.time() - total_start_time\n        \n        print_section(\"📋 SUMMARY REPORT\")\n        \n        # Find best individual model\n        best_model_name = max(results.keys(), key=lambda k: results[k]['accuracy'])\n        best_accuracy = results[best_model_name]['accuracy']\n        \n        summary_text = f\"\"\"\n🎯 BirdCLEF Neural Network Training Completed!\n\n⏱️  Total Time: {total_time:.2f} seconds ({total_time/60:.1f} minutes)\n\n📊 Data Information:\n• Total sample count: {len(train_df)}\n• Number of classes: {CFG.num_classes}\n• Feature dimensions: {X.shape[1]}\n• Train/Test: {len(y_train)}/{len(y_test)}\n\n🏆 Best Individual Model: {best_model_name}\n• Test Accuracy: {best_accuracy:.2f}%\n• Precision: {results[best_model_name]['precision']:.4f}\n• Recall: {results[best_model_name]['recall']:.4f}\n• F1-Score: {results[best_model_name]['f1']:.4f}\n\n🤖 Neural Network Architectures Trained: {len(results)}\n\"\"\"\n        \n        for arch_name, result in results.items():\n            summary_text += f\"• {arch_name}: {result['accuracy']:.2f}% accuracy\\n\"\n        \n        if ensemble_results:\n            summary_text += f\"\"\"\n🎯 Ensemble Model:\n• Method: {CFG.ensemble_method}\n• Accuracy: {ensemble_results['accuracy']:.4f}\n• Precision: {ensemble_results['precision']:.4f}\n• Recall: {ensemble_results['recall']:.4f}\n• F1-Score: {ensemble_results['f1']:.4f}\n\"\"\"\n        \n        summary_text += f\"\"\"\n📁 Saved Files:\n• Label Encoder: label_encoder.joblib\n• Individual Models: {', '.join([f'{name.lower()}_model.pth' for name in models.keys()])}\n• Training Log: training_log.log\n\"\"\"\n        \n        if ensemble:\n            summary_text += f\"• Ensemble Model: ensemble_{CFG.ensemble_method}.pkl\\n\"\n        \n        summary_text += f\"\"\"\n⚙️ Configuration Used:\n• Data Augmentation: {'✓ Enabled' if CFG.enable_data_augmentation else '✗ Disabled'}\n• Quality Filtering: {'✓ Enabled' if CFG.filter_low_quality else '✗ Disabled'}\n• Device: {device}\n• Batch Size: {CFG.batch_size}\n• Learning Rate: {CFG.learning_rate}\n        \"\"\"\n        \n        print(summary_text)\n        print(\"🎉 Pipeline completed successfully!\")\n        \n    except Exception as e:\n        print(f\"❌ Pipeline error: {e}\")\n        raise\n\nif __name__ == \"__main__\":\n    main() ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null}]}