{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Key Enhancements from Original Code\n- **Energy-based segment extraction**: Instead of extracting center 5-second segments, this implementation calculates energy distribution across the audio and selects the highest energy 5-second window, capturing the most acoustically active periods where bird/animal calls are most likely to occur.\n- **Human voice detection and removal**: Automatically detects and attenuates human voice frequencies (300-3500Hz range) commonly found in field recordings, reducing noise contamination from researchers' voices during data collection.\n- **GPU-accelerated batch processing**: Optimized pipeline using PyTorch transforms on GPU for faster mel spectrogram computation and parallel processing of multiple audio files simultaneously.\n\n## Performance Results\nUsing energy-based segment extraction for mel spectrograms significantly improves model performance compared to random 5-second sampling. The mel spectrograms generated by this notebook achieved **Private LB: 0.847** and **Public LB: 0.842** with EfficientNet-B0 5-fold cross-validation.\n\nThe energy-based 5-second segment extraction technique developed here has potential applications for future BirdCLEF competitions, offering a more intelligent approach to audio sampling that focuses on the most acoustically relevant portions of recordings.\n\n## References\n- https://www.kaggle.com/code/kadircandrisolu/transforming-audio-to-mel-spec-birdclef-25\n- https://www.kaggle.com/code/kadircandrisolu/efficientnet-b0-pytorch-train-birdclef-25/notebook\n- https://www.kaggle.com/code/kadircandrisolu/efficientnet-b0-pytorch-inference-birdclef-25?scriptVersionId=228062496\n- https://www.kaggle.com/competitions/birdclef-2025/discussion/567551 \n- https://www.kaggle.com/code/kdmitrie/bc25-separation-voice-from-data/notebook\n\n## If you find this notebook useful, please upvote this and the original notebooks!","metadata":{}},{"cell_type":"markdown","source":"## Initial Setup and Imports","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport math\nimport time\nimport librosa\nimport pandas as pd\nimport numpy as np\nimport torch\nimport torchaudio\nimport torchaudio.transforms as T\nfrom tqdm.notebook import tqdm\nimport matplotlib.pyplot as plt\nimport warnings\nfrom concurrent.futures import ThreadPoolExecutor\nwarnings.filterwarnings(\"ignore\")\n\n# GPU/CPU configuration\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(f\"Using device: {device}\")\nif torch.cuda.is_available():\n    print(f\"GPU: {torch.cuda.get_device_name(0)}\")\n    print(f\"VRAM: {torch.cuda.get_device_properties(0).total_memory / 1e9:.1f} GB\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T02:40:16.266916Z","iopub.execute_input":"2025-06-03T02:40:16.267246Z","iopub.status.idle":"2025-06-03T02:40:16.274969Z","shell.execute_reply.started":"2025-06-03T02:40:16.267218Z","shell.execute_reply":"2025-06-03T02:40:16.274041Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Configuration Class","metadata":{}},{"cell_type":"code","source":"class OptimizedConfig:\n    DEBUG_MODE = True\n    \n    OUTPUT_DIR = '/kaggle/working/'\n    DATA_ROOT = '/kaggle/input/birdclef-2025'\n    FS = 32000\n    \n    # My Mel spectrogram parameters \n    N_FFT = 1024\n    HOP_LENGTH = 160\n    N_MELS = 100\n    FMIN = 20\n    FMAX = 16000\n    NORMALIZED = True  \n    TOP_DB = 80\n    \n    TARGET_DURATION = 5.0\n    TARGET_SHAPE = (256, 256)\n    \n    # Audio extraction method flag: True for energy-based, False for center extraction\n    USE_ENERGY_EXTRACTION = True\n    \n    # Human voice detection parameters\n    VOICE_CHUNK_LEN = 0.05\n    VOICE_ENERGY_THRESHOLD = -50\n    VOICE_FREQ_RANGE = (300, 3500)\n    \n    # Energy calculation parameters\n    ENERGY_WINDOW_SIZE = 0.1\n    ENERGY_HOP_SIZE = 0.05\n    \n    # Batch processing parameters\n    BATCH_SIZE = 8\n    PREFETCH_SIZE = 4\n    \n    N_MAX = 1000 if DEBUG_MODE else None\n\nconfig = OptimizedConfig()\nprint(f\"Debug mode: {'ON' if config.DEBUG_MODE else 'OFF'}\")\nprint(f\"Max samples to process: {config.N_MAX if config.N_MAX is not None else 'ALL'}\")\nprint(f\"Audio extraction method: {'Energy-based' if config.USE_ENERGY_EXTRACTION else 'Center-based'}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T02:40:16.276181Z","iopub.execute_input":"2025-06-03T02:40:16.276437Z","iopub.status.idle":"2025-06-03T02:40:16.301091Z","shell.execute_reply.started":"2025-06-03T02:40:16.276415Z","shell.execute_reply":"2025-06-03T02:40:16.300186Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Loading","metadata":{}},{"cell_type":"code","source":"# Load taxonomy data\nprint(\"Loading taxonomy data...\")\ntaxonomy_df = pd.read_csv(f'{config.DATA_ROOT}/taxonomy.csv')\nspecies_class_map = dict(zip(taxonomy_df['primary_label'], taxonomy_df['class_name']))\n\nprint(\"Loading training metadata...\")\ntrain_df = pd.read_csv(f'{config.DATA_ROOT}/train.csv')\nlabel_list = sorted(train_df['primary_label'].unique())\nlabel_id_list = list(range(len(label_list)))\nlabel2id = dict(zip(label_list, label_id_list))\nid2label = dict(zip(label_id_list, label_list))\nprint(f'Found {len(label_list)} unique species')\n\nworking_df = train_df[['primary_label', 'rating', 'filename']].copy()\nworking_df['target'] = working_df.primary_label.map(label2id)\nworking_df['filepath'] = config.DATA_ROOT + '/train_audio/' + working_df.filename\nworking_df['samplename'] = working_df.filename.map(lambda x: x.split('/')[0] + '-' + x.split('/')[-1].split('.')[0])\nworking_df['class'] = working_df.primary_label.map(lambda x: species_class_map.get(x, 'Unknown'))\n\ntotal_samples = min(len(working_df), config.N_MAX or len(working_df))\nprint(f'Total samples to process: {total_samples} out of {len(working_df)} available')\nprint(f'Samples by class:')\nprint(working_df['class'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T02:40:16.302268Z","iopub.execute_input":"2025-06-03T02:40:16.302558Z","iopub.status.idle":"2025-06-03T02:40:16.554867Z","shell.execute_reply.started":"2025-06-03T02:40:16.302516Z","shell.execute_reply":"2025-06-03T02:40:16.554022Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## GPU Audio Processor Class","metadata":{}},{"cell_type":"code","source":"class GPUAudioProcessor:\n    def __init__(self, config, device):\n        self.config = config\n        self.device = device\n        \n        # Pre-create transforms and move to GPU\n        self.mel_transform = T.MelSpectrogram(\n            sample_rate=config.FS,\n            n_fft=config.N_FFT,\n            hop_length=config.HOP_LENGTH,\n            n_mels=config.N_MELS,\n            f_min=config.FMIN,\n            f_max=config.FMAX,\n            norm='slaney' if config.NORMALIZED else None,\n            mel_scale='slaney'\n        ).to(device)\n        \n        self.amplitude_to_db = T.AmplitudeToDB(\n            stype='power',\n            top_db=config.TOP_DB\n        ).to(device)\n        \n        # Pre-calculate frequency bin indices\n        self.freqs = librosa.fft_frequencies(sr=config.FS, n_fft=config.N_FFT)\n        self.voice_low_idx = np.argmin(np.abs(self.freqs - config.VOICE_FREQ_RANGE[0]))\n        self.voice_high_idx = np.argmin(np.abs(self.freqs - config.VOICE_FREQ_RANGE[1]))\n    \n    def extract_center_segments_batch(self, audio_batch):\n        \"\"\"Extract center segments from audio batch (traditional method)\"\"\"\n        # Convert 1D to 2D if necessary\n        if len(audio_batch.shape) == 1:\n            audio_batch = audio_batch.reshape(1, -1)\n        \n        batch_size = audio_batch.shape[0]\n        target_samples = int(self.config.TARGET_DURATION * self.config.FS)\n        \n        segments = []\n        start_indices = np.zeros(batch_size, dtype=int)\n        \n        for i in range(batch_size):\n            audio_data = audio_batch[i]\n            audio_length = len(audio_data)\n            \n            # If audio is too short, use the entire audio\n            if audio_length <= target_samples:\n                segment = audio_data\n                if len(segment) < target_samples:\n                    segment = np.pad(segment, (0, target_samples - len(segment)), mode='constant')\n                segments.append(segment)\n                start_indices[i] = 0\n                continue\n            \n            # Extract center segment\n            start_idx = max(0, int(audio_length / 2 - target_samples / 2))\n            end_idx = start_idx + target_samples\n            segment = audio_data[start_idx:end_idx]\n            \n            segments.append(segment)\n            start_indices[i] = start_idx\n        \n        return segments, start_indices\n    \n    def find_highest_energy_segments_batch(self, audio_batch):\n        \"\"\"Batch processing for highest energy segment detection\"\"\"\n        # Convert 1D to 2D if necessary\n        if len(audio_batch.shape) == 1:\n            audio_batch = audio_batch.reshape(1, -1)\n        \n        batch_size = audio_batch.shape[0]\n        target_samples = int(self.config.TARGET_DURATION * self.config.FS)\n        \n        segments = []\n        start_indices = np.zeros(batch_size, dtype=int)\n        \n        for i in range(batch_size):\n            audio_data = audio_batch[i]\n            audio_length = len(audio_data)\n            \n            # If audio is too short, use the entire audio\n            if audio_length <= target_samples:\n                segment = audio_data\n                if len(segment) < target_samples:\n                    segment = np.pad(segment, (0, target_samples - len(segment)), mode='constant')\n                segments.append(segment)\n                start_indices[i] = 0\n                continue\n            \n            # GPU-based energy calculation\n            audio_tensor = torch.from_numpy(audio_data).float().to(self.device)\n            \n            window_samples = int(self.config.ENERGY_WINDOW_SIZE * self.config.FS)\n            hop_samples = int(self.config.ENERGY_HOP_SIZE * self.config.FS)\n            \n            # Sliding window processing\n            n_windows = 1 + int((audio_length - window_samples) / hop_samples)\n            \n            if n_windows <= 0:\n                # Audio is too short\n                start_idx = 0\n                segment = audio_data[start_idx:start_idx + target_samples]\n                if len(segment) < target_samples:\n                    segment = np.pad(segment, (0, target_samples - len(segment)), mode='constant')\n                segments.append(segment)\n                start_indices[i] = start_idx\n                continue\n            \n            try:\n                # Energy calculation\n                unfolded = audio_tensor.unfold(0, window_samples, hop_samples)\n                energies = (unfolded ** 2).sum(dim=1)\n                \n                # Calculate average energy for 5-second windows\n                segment_windows = int(self.config.TARGET_DURATION / self.config.ENERGY_HOP_SIZE)\n                \n                if segment_windows >= n_windows:\n                    # Segment is too long, extract from center\n                    start_idx = max(0, int(audio_length / 2 - target_samples / 2))\n                else:\n                    # Convolution for moving average\n                    kernel = torch.ones(segment_windows).to(self.device) / segment_windows\n                    avg_energies = torch.nn.functional.conv1d(\n                        energies.unsqueeze(0).unsqueeze(0),\n                        kernel.unsqueeze(0).unsqueeze(0),\n                        padding=0\n                    ).squeeze()\n                    \n                    # Find maximum energy position\n                    max_idx = avg_energies.argmax().item()\n                    start_idx = max_idx * hop_samples\n                \n                # Extract segment\n                start_idx = min(start_idx, audio_length - target_samples)\n                end_idx = start_idx + target_samples\n                segment = audio_data[start_idx:end_idx]\n                \n                segments.append(segment)\n                start_indices[i] = start_idx\n                \n            except Exception as e:\n                print(f\"Error in energy segment detection: {e}\")\n                # Fallback: extract from center\n                start_idx = max(0, int(audio_length / 2 - target_samples / 2))\n                segment = audio_data[start_idx:start_idx + target_samples]\n                if len(segment) < target_samples:\n                    segment = np.pad(segment, (0, target_samples - len(segment)), mode='constant')\n                segments.append(segment)\n                start_indices[i] = start_idx\n        \n        return segments, start_indices\n    \n    def detect_and_remove_voice_batch(self, audio_segments):\n        \"\"\"Batch processing for human voice detection and removal\"\"\"\n        batch_segments = []\n        \n        for idx, audio in enumerate(audio_segments):\n            try:\n                # GPU-based STFT\n                audio_tensor = torch.from_numpy(audio).float().to(self.device)\n                stft = torch.stft(audio_tensor, \n                                 n_fft=self.config.N_FFT, \n                                 hop_length=self.config.HOP_LENGTH,\n                                 return_complex=True)\n                \n                # Energy-based voice detection\n                chunk_samples = int(self.config.VOICE_CHUNK_LEN * self.config.FS)\n                audio_power = audio_tensor ** 2\n                \n                # Calculate energy per chunk\n                n_chunks = int(np.ceil(len(audio) / chunk_samples))\n                if n_chunks == 0:\n                    batch_segments.append(audio)\n                    continue\n                    \n                pad_size = n_chunks * chunk_samples - len(audio)\n                if pad_size > 0:\n                    audio_power = torch.nn.functional.pad(audio_power, (0, pad_size))\n                \n                power_chunks = audio_power.view(n_chunks, chunk_samples).sum(dim=1)\n                power_db = 10 * torch.log10(power_chunks + 1e-10)\n                \n                # Detect intersections with threshold\n                x = power_db - self.config.VOICE_ENERGY_THRESHOLD\n                x_cpu = x.cpu().numpy()\n                \n                # Detect intersections with numpy\n                if len(x_cpu) > 1:\n                    intersections = np.where(x_cpu[:-1] * x_cpu[1:] < 0)[0]\n                else:\n                    intersections = np.array([])\n                \n                if len(intersections) >= 2:\n                    # Detect human voice segments\n                    start = int(intersections[0] * self.config.VOICE_CHUNK_LEN * self.config.FS)\n                    end = int(intersections[1] * self.config.VOICE_CHUNK_LEN * self.config.FS)\n                    \n                    # Check STFT shape and access\n                    start_frame = max(0, start // self.config.HOP_LENGTH)\n                    end_frame = min(stft.shape[1], end // self.config.HOP_LENGTH)\n                    \n                    # Apply mask\n                    mask = torch.ones_like(stft)\n                    mask[self.voice_low_idx:self.voice_high_idx, start_frame:end_frame] *= 0.2\n                    filtered_stft = stft * mask\n                    \n                    # Inverse transform\n                    filtered_audio = torch.istft(filtered_stft, \n                                               n_fft=self.config.N_FFT, \n                                               hop_length=self.config.HOP_LENGTH,\n                                               length=len(audio))\n                    \n                else:\n                    filtered_audio = audio_tensor\n                \n                batch_segments.append(filtered_audio.cpu().numpy())\n                \n            except Exception as e:\n                print(f\"Error in voice removal for segment {idx}: {e}\")\n                batch_segments.append(audio)\n        \n        return batch_segments\n    \n    def compute_mel_spectrograms_batch(self, audio_segments):\n        \"\"\"Batch processing for mel spectrogram computation\"\"\"\n        try:\n            # Create batch tensor\n            audio_batch = np.stack(audio_segments)\n            audio_tensor = torch.from_numpy(audio_batch).float().to(self.device)\n            \n            # Batch mel spectrogram computation\n            mel_specs = self.mel_transform(audio_tensor)\n            mel_specs_db = self.amplitude_to_db(mel_specs)\n            \n            # Normalization\n            mel_specs_norm = (mel_specs_db + self.config.TOP_DB) / self.config.TOP_DB\n            mel_specs_norm = torch.clamp(mel_specs_norm, 0, 1)\n            \n            # Move to CPU and resize\n            mel_specs_cpu = mel_specs_norm.cpu().numpy()\n            \n            resized_specs = []\n            for spec in mel_specs_cpu:\n                if spec.shape != self.config.TARGET_SHAPE:\n                    spec_resized = cv2.resize(spec, \n                                            (self.config.TARGET_SHAPE[1], self.config.TARGET_SHAPE[0]), \n                                            interpolation=cv2.INTER_LINEAR)\n                else:\n                    spec_resized = spec\n                resized_specs.append(spec_resized)\n            \n            return resized_specs\n        except Exception as e:\n            print(f\"Error in mel spectrogram computation: {e}\")\n            raise\n\n# Initialize GPU audio processor\naudio_processor = GPUAudioProcessor(config, device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T02:40:16.555800Z","iopub.execute_input":"2025-06-03T02:40:16.556047Z","iopub.status.idle":"2025-06-03T02:40:18.013374Z","shell.execute_reply.started":"2025-06-03T02:40:16.556013Z","shell.execute_reply":"2025-06-03T02:40:18.012461Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Batch Processing Function","metadata":{}},{"cell_type":"code","source":"# Batch processing data loader\ndef process_batch(df_batch):\n    \"\"\"Batch processing function (modified version)\"\"\"\n    filepaths = df_batch['filepath'].tolist()\n    target_samples = int(config.TARGET_DURATION * config.FS)\n    \n    # Batch loading\n    audio_batch = []\n    valid_indices = []\n    original_lengths = []\n    \n    for i, filepath in enumerate(filepaths):\n        try:\n            audio, _ = librosa.load(filepath, sr=config.FS)\n            original_length = len(audio)\n            \n            # Keep audio length (don't truncate)\n            audio_batch.append(audio)\n            valid_indices.append(i)\n            original_lengths.append(original_length)\n        except Exception as e:\n            print(f\"Error loading {filepath}: {e}\")\n            continue\n    \n    if not audio_batch:\n        return []\n    \n    # Padding to max length\n    max_length = max(len(audio) for audio in audio_batch)\n    padded_batch = []\n    for audio in audio_batch:\n        if len(audio) < max_length:\n            padded = np.pad(audio, (0, max_length - len(audio)), mode='constant')\n        else:\n            padded = audio\n        padded_batch.append(padded)\n    \n    # Convert to numpy array\n    audio_array = np.array(padded_batch, dtype=np.float32)\n    \n    # Extract segments based on configuration flag\n    if config.USE_ENERGY_EXTRACTION:\n        segments, start_indices = audio_processor.find_highest_energy_segments_batch(audio_array)\n        extraction_method = \"energy\"\n    else:\n        segments, start_indices = audio_processor.extract_center_segments_batch(audio_array)\n        extraction_method = \"center\"\n    \n    # Human voice removal\n    filtered_segments = audio_processor.detect_and_remove_voice_batch(segments)\n    \n    # Mel spectrogram computation\n    mel_specs = audio_processor.compute_mel_spectrograms_batch(filtered_segments)\n    \n    # Collect results\n    results = []\n    for i, idx in enumerate(valid_indices):\n        row = df_batch.iloc[idx]\n        results.append({\n            'samplename': row['samplename'],\n            'mel_spec': mel_specs[i],\n            'start_time': start_indices[i] / config.FS,\n            'end_time': (start_indices[i] + target_samples) / config.FS,\n            'total_duration': original_lengths[i] / config.FS,\n            'filename': row['filename'],\n            'voice_detected': not np.array_equal(filtered_segments[i], segments[i]),\n            'extraction_method': extraction_method\n        })\n    \n    return results","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T02:40:18.015289Z","iopub.execute_input":"2025-06-03T02:40:18.015511Z","iopub.status.idle":"2025-06-03T02:40:18.024186Z","shell.execute_reply.started":"2025-06-03T02:40:18.015491Z","shell.execute_reply":"2025-06-03T02:40:18.023501Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Main Processing Loop","metadata":{}},{"cell_type":"code","source":"# Execute batch processing\nall_bird_data = {}\nerrors = []\nvoice_detected_count = 0\nhighest_energy_segments = []\n\nprint(\"Starting optimized audio processing with GPU acceleration...\")\nprint(f\"{'DEBUG MODE - Processing only 1000 samples' if config.DEBUG_MODE else 'FULL MODE - Processing all samples'}\")\n\nstart_time = time.time()\n\nbatch_indices = range(0, total_samples, config.BATCH_SIZE)\nfor i in tqdm(batch_indices):\n    batch_df = working_df.iloc[i:min(i + config.BATCH_SIZE, total_samples)]\n    \n    try:\n        batch_results = process_batch(batch_df)\n        \n        for result in batch_results:\n            all_bird_data[result['samplename']] = result['mel_spec'].astype(np.float32)\n            \n            highest_energy_segments.append({\n                'filename': result['filename'],\n                'start_time': result['start_time'],\n                'end_time': result['end_time'],\n                'total_duration': result['total_duration'],\n                'extraction_method': result['extraction_method']\n            })\n            \n            if result['voice_detected']:\n                voice_detected_count += 1\n    except Exception as e:\n        print(f\"Error processing batch {i//config.BATCH_SIZE}: {e}\")\n        errors.append((i, str(e)))\n    \n    # Memory management: clear GPU memory\n    if torch.cuda.is_available():\n        torch.cuda.empty_cache()\n\nend_time = time.time()\nprint(f\"Processing completed in {end_time - start_time:.2f} seconds\")\nprint(f\"Successfully processed {len(all_bird_data)} files out of {total_samples} total\")\nprint(f\"Detected and removed human voice in {voice_detected_count} files\")\nprint(f\"Failed to process {len(errors)} batches\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T02:46:52.702502Z","iopub.execute_input":"2025-06-03T02:46:52.702877Z","iopub.status.idle":"2025-06-03T02:47:52.878424Z","shell.execute_reply.started":"2025-06-03T02:46:52.702849Z","shell.execute_reply":"2025-06-03T02:47:52.877327Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Saving","metadata":{}},{"cell_type":"code","source":"# Save mel spectrogram data as .npy file\nextraction_suffix = \"energy\" if config.USE_ENERGY_EXTRACTION else \"center\"\ndebug_suffix = \"_debug\" if config.DEBUG_MODE else \"\"\noutput_file = f\"{config.OUTPUT_DIR}/optimized_melspecs_data_{extraction_suffix}{debug_suffix}.npy\"\nnp.save(output_file, all_bird_data)\nprint(f\"Mel spectrogram data saved to {output_file}\")\n\n# Save segment information\nsegments_output = f\"{config.OUTPUT_DIR}/optimized_{extraction_suffix}_segments{debug_suffix}.csv\"\nif highest_energy_segments:\n    pd.DataFrame(highest_energy_segments).to_csv(segments_output, index=False)\n    print(f\"Segment information saved to {segments_output}\")\n\n# Display energy distribution statistics\nif highest_energy_segments:\n    df_segments = pd.DataFrame(highest_energy_segments)\n    relative_positions = df_segments['start_time'] / df_segments['total_duration']\n    \n    print(f\"\\n{extraction_suffix.capitalize()} segment statistics:\")\n    print(f\"Average relative position: {relative_positions.mean():.3f}\")\n    print(f\"Median relative position: {relative_positions.median():.3f}\")\n    print(f\"Segments in first third: {(relative_positions < 0.33).sum()} ({(relative_positions < 0.33).mean():.1%})\")\n    print(f\"Segments in middle third: {((relative_positions >= 0.33) & (relative_positions < 0.67)).sum()} ({((relative_positions >= 0.33) & (relative_positions < 0.67)).mean():.1%})\")\n    print(f\"Segments in last third: {(relative_positions >= 0.67).sum()} ({(relative_positions >= 0.67).mean():.1%})\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T02:47:52.879603Z","iopub.execute_input":"2025-06-03T02:47:52.879830Z","iopub.status.idle":"2025-06-03T02:47:53.219152Z","shell.execute_reply.started":"2025-06-03T02:47:52.879810Z","shell.execute_reply":"2025-06-03T02:47:53.218230Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Visualization Functions","metadata":{}},{"cell_type":"code","source":"# English font configuration for matplotlib\nplt.rcParams['axes.unicode_minus'] = False\n\ndef plot_comparison_example(audio_file, fs=32000):\n    \"\"\"Visualization comparing center extraction and energy-based extraction\"\"\"\n    audio_data, _ = librosa.load(audio_file, sr=fs)\n    \n    # Target sample count\n    target_samples = int(config.TARGET_DURATION * fs)\n    \n    # 1. Extract center 5 seconds (original code)\n    start_idx_center = max(0, int(len(audio_data) / 2 - target_samples / 2))\n    end_idx_center = min(len(audio_data), start_idx_center + target_samples)\n    center_audio = audio_data[start_idx_center:end_idx_center]\n    \n    # 2. Extract highest energy 5 seconds (new code)\n    segments, start_indices = audio_processor.find_highest_energy_segments_batch(\n        np.array([audio_data])\n    )\n    energy_audio = segments[0]\n    start_idx_energy = start_indices[0]\n    end_idx_energy = min(len(audio_data), start_idx_energy + target_samples)\n    \n    # Padding if necessary\n    if len(center_audio) < target_samples:\n        center_audio = np.pad(center_audio, (0, target_samples - len(center_audio)), mode='constant')\n    if len(energy_audio) < target_samples:\n        energy_audio = np.pad(energy_audio, (0, target_samples - len(energy_audio)), mode='constant')\n    \n    # Human voice removal\n    filtered_center_audio = audio_processor.detect_and_remove_voice_batch([center_audio])[0]\n    filtered_energy_audio = audio_processor.detect_and_remove_voice_batch([energy_audio])[0]\n    \n    # Mel spectrograms\n    mel_center = audio_processor.compute_mel_spectrograms_batch([filtered_center_audio])[0]\n    mel_energy = audio_processor.compute_mel_spectrograms_batch([filtered_energy_audio])[0]\n    \n    # Visualize waveforms and energy selection regions\n    fig, ax = plt.subplots(3, 2, figsize=(18, 12))\n    \n    # Display entire waveform and energy selection regions\n    time_axis = np.arange(len(audio_data)) / fs\n    ax[0, 0].plot(time_axis, audio_data)\n    ax[0, 0].axvspan(start_idx_center/fs, end_idx_center/fs, color='r', alpha=0.3, label='Center region')\n    ax[0, 0].axvspan(start_idx_energy/fs, end_idx_energy/fs, color='g', alpha=0.3, label='Highest energy region')\n    ax[0, 0].set_title('Full Audio and Extraction Regions')\n    ax[0, 0].set_xlabel('Time (seconds)')\n    ax[0, 0].legend()\n    \n    # Compare extracted waveforms\n    ax[0, 1].plot(np.arange(len(center_audio))/fs, center_audio, 'r-', label='Center region')\n    ax[0, 1].plot(np.arange(len(energy_audio))/fs, energy_audio, 'g-', label='Highest energy region')\n    ax[0, 1].set_title('Comparison of Extracted Waveforms')\n    ax[0, 1].set_xlabel('Time (seconds)')\n    ax[0, 1].legend()\n    \n    # Center region mel spectrogram (original and after voice removal)\n    mel_center_orig = audio_processor.compute_mel_spectrograms_batch([center_audio])[0]\n    ax[1, 0].imshow(mel_center_orig, aspect='auto', origin='lower', cmap='viridis')\n    ax[1, 0].set_title('Center Region Mel Spectrogram')\n    \n    ax[1, 1].imshow(mel_center, aspect='auto', origin='lower', cmap='viridis')\n    ax[1, 1].set_title('Center Region (After Voice Removal)')\n    \n    # Highest energy region mel spectrogram (original and after voice removal)\n    mel_energy_orig = audio_processor.compute_mel_spectrograms_batch([energy_audio])[0]\n    ax[2, 0].imshow(mel_energy_orig, aspect='auto', origin='lower', cmap='viridis')\n    ax[2, 0].set_title('Highest Energy Region Mel Spectrogram')\n    \n    ax[2, 1].imshow(mel_energy, aspect='auto', origin='lower', cmap='viridis')\n    ax[2, 1].set_title('Highest Energy Region (After Voice Removal)')\n    \n    plt.tight_layout()\n    return fig\n\ndef plot_voice_removal_example(audio_file, fs=32000):\n    \"\"\"Visualization showing the effect of human voice removal\"\"\"\n    audio_data, _ = librosa.load(audio_file, sr=fs)\n    \n    # Extract based on configuration\n    if config.USE_ENERGY_EXTRACTION:\n        segments, start_indices = audio_processor.find_highest_energy_segments_batch(\n            np.array([audio_data])\n        )\n    else:\n        segments, start_indices = audio_processor.extract_center_segments_batch(\n            np.array([audio_data])\n        )\n    selected_audio = segments[0]\n    \n    # Human voice removal\n    filtered_audio = audio_processor.detect_and_remove_voice_batch([selected_audio])[0]\n    \n    # Mel spectrograms\n    mel_orig = audio_processor.compute_mel_spectrograms_batch([selected_audio])[0]\n    mel_filtered = audio_processor.compute_mel_spectrograms_batch([filtered_audio])[0]\n    \n    # Visualization\n    fig, ax = plt.subplots(2, 2, figsize=(14, 10))\n    \n    # Waveforms\n    method_name = \"Highest Energy\" if config.USE_ENERGY_EXTRACTION else \"Center\"\n    ax[0, 0].plot(np.arange(len(selected_audio))/fs, selected_audio)\n    ax[0, 0].set_title(f'Original Audio Waveform ({method_name} Region)')\n    ax[0, 0].set_xlabel('Time (seconds)')\n    \n    ax[0, 1].plot(np.arange(len(filtered_audio))/fs, filtered_audio)\n    ax[0, 1].set_title('Waveform After Voice Removal')\n    ax[0, 1].set_xlabel('Time (seconds)')\n    \n    # Mel spectrograms\n    ax[1, 0].imshow(mel_orig, aspect='auto', origin='lower', cmap='viridis')\n    ax[1, 0].set_title('Original Audio Mel Spectrogram')\n    \n    ax[1, 1].imshow(mel_filtered, aspect='auto', origin='lower', cmap='viridis')\n    ax[1, 1].set_title('Mel Spectrogram After Voice Removal')\n    \n    plt.tight_layout()\n    return fig","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T02:47:53.220968Z","iopub.execute_input":"2025-06-03T02:47:53.221208Z","iopub.status.idle":"2025-06-03T02:47:53.235253Z","shell.execute_reply.started":"2025-06-03T02:47:53.221188Z","shell.execute_reply":"2025-06-03T02:47:53.234392Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Sample Visualization","metadata":{}},{"cell_type":"code","source":"# Sample visualization\nsamples = []\ndisplayed_classes = set()\nmax_samples = min(4, len(all_bird_data))\nfor i, row in working_df.iterrows():\n    if i >= (config.N_MAX or len(working_df)):\n        break\n        \n    if row['samplename'] in all_bird_data:\n        if row['class'] not in displayed_classes:\n            samples.append((row['samplename'], row['class'], row['primary_label']))\n            displayed_classes.add(row['class'])\n        \n        if len(samples) >= max_samples:  \n            break\n\nif samples:\n    plt.figure(figsize=(16, 12))\n    \n    method_name = \"High Energy\" if config.USE_ENERGY_EXTRACTION else \"Center\"\n    for i, (samplename, class_name, species) in enumerate(samples):\n        plt.subplot(2, 2, i+1)\n        plt.imshow(all_bird_data[samplename], aspect='auto', origin='lower', cmap='viridis')\n        plt.title(f\"{class_name}: {species} ({method_name} Segment)\")\n        plt.colorbar(format='%+2.0f dB')\n    \n    plt.tight_layout()\n    debug_note = \"optimized_debug_\" if config.DEBUG_MODE else \"optimized_\"\n    method_suffix = \"energy_\" if config.USE_ENERGY_EXTRACTION else \"center_\"\n    plt.savefig(f'{config.OUTPUT_DIR}/{debug_note}{method_suffix}melspec_examples.png')\n    plt.show()\n    \n    print(f\"Saved visualization to {config.OUTPUT_DIR}/{debug_note}{method_suffix}melspec_examples.png\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T02:47:53.236339Z","iopub.execute_input":"2025-06-03T02:47:53.236647Z","iopub.status.idle":"2025-06-03T02:47:56.315286Z","shell.execute_reply.started":"2025-06-03T02:47:53.236623Z","shell.execute_reply":"2025-06-03T02:47:56.313950Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Comparison and Analysis Visualizations","metadata":{}},{"cell_type":"code","source":"# Comparison visualization between center and highest energy segments\nsample_file = None\nfor i, row in working_df.iloc[:min(20, len(working_df))].iterrows():\n    file_path = f\"{config.DATA_ROOT}/train_audio/{row.filename}\"\n    if os.path.exists(file_path):\n        sample_file = file_path\n        break\n\nif sample_file and config.USE_ENERGY_EXTRACTION:\n    print(f\"Visualizing comparison between center and highest energy segments for: {os.path.basename(sample_file)}\")\n    fig = plot_comparison_example(sample_file)\n    plt.savefig(f'{config.OUTPUT_DIR}/optimized_center_vs_energy_comparison.png', bbox_inches='tight')\n    plt.show()\n    print(f\"Saved comparison visualization to {config.OUTPUT_DIR}/optimized_center_vs_energy_comparison.png\")\n\n# Voice removal effect visualization on Fabio's recording samples\nfabio_authors = train_df[train_df.author == 'Fabio A. Sarria-S']\nif len(fabio_authors) > 0:\n    sample_file = f\"{config.DATA_ROOT}/train_audio/{fabio_authors.iloc[0].filename}\"\n    print(f\"Visualizing voice removal effect on: {sample_file}\")\n    fig = plot_voice_removal_example(sample_file)\n    method_suffix = \"energy_\" if config.USE_ENERGY_EXTRACTION else \"center_\"\n    plt.savefig(f'{config.OUTPUT_DIR}/optimized_{method_suffix}voice_removal_example.png', bbox_inches='tight')\n    plt.show()\n    print(f\"Saved example visualization to {config.OUTPUT_DIR}/optimized_{method_suffix}voice_removal_example.png\")\n\n# Energy distribution visualization\nif highest_energy_segments:\n    df_segments = pd.DataFrame(highest_energy_segments)\n    relative_positions = df_segments['start_time'] / df_segments['total_duration']\n    \n    plt.figure(figsize=(12, 6))\n    plt.hist(relative_positions, bins=20, color='skyblue', edgecolor='black')\n    plt.axvline(x=relative_positions.mean(), color='red', linestyle='--', \n                label=f'Mean: {relative_positions.mean():.3f}')\n    plt.axvline(x=relative_positions.median(), color='green', linestyle='--', \n                label=f'Median: {relative_positions.median():.3f}')\n    plt.axvline(x=0.5, color='black', linestyle=':', label='Audio Center Point')\n    plt.xlabel('Relative Position in Audio (0=start, 1=end)')\n    plt.ylabel('Number of Samples')\n    \n    method_name = \"Highest Energy\" if config.USE_ENERGY_EXTRACTION else \"Center\"\n    plt.title(f'{method_name} Segment Distribution in Audio')\n    plt.legend()\n    plt.grid(alpha=0.3)\n    plt.tight_layout()\n    \n    method_suffix = \"energy_\" if config.USE_ENERGY_EXTRACTION else \"center_\"\n    plt.savefig(f'{config.OUTPUT_DIR}/optimized_{method_suffix}segment_distribution.png', dpi=300, bbox_inches='tight')\n    plt.show()\n    print(f\"Saved segment distribution visualization to {config.OUTPUT_DIR}/optimized_{method_suffix}segment_distribution.png\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T02:47:56.316376Z","iopub.execute_input":"2025-06-03T02:47:56.316654Z","iopub.status.idle":"2025-06-03T02:48:28.743171Z","shell.execute_reply.started":"2025-06-03T02:47:56.316629Z","shell.execute_reply":"2025-06-03T02:48:28.742177Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Loading Test","metadata":{}},{"cell_type":"code","source":"# Test loading saved data\ndef test_saved_data():\n    print(\"\\nTesting data loading from saved file...\")\n    try:\n        loaded_data = np.load(output_file, allow_pickle=True).item()\n        print(f\"Successfully loaded {len(loaded_data)} mel spectrograms\")\n        print(f\"First key: {list(loaded_data.keys())[0]}\")\n        print(f\"Shape of first spectrogram: {loaded_data[list(loaded_data.keys())[0]].shape}\")\n        return True\n    except Exception as e:\n        print(f\"Error loading saved data: {e}\")\n        return False\n\n# Execute test\ntest_saved_data()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-03T02:48:28.743979Z","iopub.execute_input":"2025-06-03T02:48:28.744262Z","iopub.status.idle":"2025-06-03T02:48:28.914285Z","shell.execute_reply.started":"2025-06-03T02:48:28.744239Z","shell.execute_reply":"2025-06-03T02:48:28.913583Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}