{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30920,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### What is it?\n\nThis code will create **5 second** 224x224 PNG melspectrograms for BirdCLEF+ 2025. *Runtime about 5 hours.\n- Please find output of this code at the [BirdCLEF+ 2025, MelSPECs (5 sec) Dataset](https://www.kaggle.com/datasets/samvelkoch/birdclef-2025-melspecs-5-sec/)\n- Addon: **10 seconds** Melspectrograms [Code](https://www.kaggle.com/code/samvelkoch/birdclef2025-dataprep-specs-10-sec/) and [Dataset](https://www.kaggle.com/datasets/samvelkoch/birdclef-2025-melspecs-10-sec)\n\nHave fun... Thank you 🙏","metadata":{"execution":{"iopub.status.busy":"2025-03-11T14:24:24.494212Z","iopub.execute_input":"2025-03-11T14:24:24.494440Z","iopub.status.idle":"2025-03-11T14:24:36.406246Z","shell.execute_reply.started":"2025-03-11T14:24:24.494416Z","shell.execute_reply":"2025-03-11T14:24:36.404824Z"}}},{"cell_type":"code","source":"# Audio Processing for BirdCLEF 2025\n# Creating mel-spectrograms from 5-second segments with parallel processing (without saving audio)\n\nimport os\nimport numpy as np\nimport librosa\nimport librosa.display\nimport matplotlib.pyplot as plt\nfrom matplotlib import cm\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom PIL import Image\nimport glob\nimport concurrent.futures\nimport multiprocessing\nimport tqdm\ntqdm.tqdm = tqdm.tqdm_notebook = tqdm.tqdm\n\n# Path configuration\nBASE_PATH = '/kaggle/input/birdclef-2025'\nTRAIN_AUDIO_PATH = os.path.join(BASE_PATH, 'train_audio')\n\n# Number of workers for parallel processing\n# Using 80% of available CPUs is usually a good balance\nNUM_WORKERS = max(1, int(multiprocessing.cpu_count() * 0.8))\nprint(f\"Using {NUM_WORKERS} workers for parallel processing\")\n\n# Create destination directories\ndef create_output_dirs():\n    # Only for spectrograms\n    os.makedirs('specs_5sec', exist_ok=True)\n    \n    print(\"Output directory created\")\n\n# Function to slice audio into segments\ndef segment_audio(audio, sr, duration):\n    \"\"\"\n    Slices audio into segments of specified duration\n    \n    Parameters:\n    -----------\n    audio : np.array\n        Audio data\n    sr : int\n        Sample rate\n    duration : int\n        Segment duration in seconds\n    \n    Returns:\n    --------\n    list\n        List of audio segments\n    \"\"\"\n    # Calculate number of samples for the desired duration\n    samples_per_segment = int(sr * duration)\n    \n    # Calculate number of full segments\n    num_segments = len(audio) // samples_per_segment\n    \n    segments = []\n    for i in range(num_segments):\n        start = i * samples_per_segment\n        end = start + samples_per_segment\n        segment = audio[start:end]\n        segments.append(segment)\n    \n    return segments\n\n# Function to create mel-spectrogram\ndef create_melspectrogram(audio, sr, size=(224, 224)):\n    \"\"\"\n    Creates a mel-spectrogram from audio data\n    \n    Parameters:\n    -----------\n    audio : np.array\n        Audio data\n    sr : int\n        Sample rate\n    size : tuple\n        Output image size (width, height)\n    \n    Returns:\n    --------\n    PIL.Image\n        Mel-spectrogram image of size (width, height)\n    \"\"\"\n    # Calculate mel-spectrogram\n    melspec = librosa.feature.melspectrogram(\n        y=audio, \n        sr=sr, \n        n_mels=128,\n        fmax=sr/2\n    )\n    \n    # Convert to decibels\n    melspec_db = librosa.power_to_db(melspec, ref=np.max)\n    \n    # Create figure of required size\n    fig, ax = plt.subplots(figsize=(10, 4))\n    \n    # Display spectrogram\n    img = librosa.display.specshow(\n        melspec_db, \n        x_axis='time', \n        y_axis='mel', \n        sr=sr,\n        cmap=cm.viridis,\n        ax=ax\n    )\n    \n    # Remove axes and padding\n    plt.axis('off')\n    plt.tight_layout(pad=0)\n    \n    # Use a unique temporary filename for parallel processing\n    import uuid\n    temp_file = f'temp_spec_{uuid.uuid4()}.png'\n    plt.savefig(temp_file, bbox_inches='tight', pad_inches=0)\n    plt.close(fig)\n    \n    # Open and resize the image\n    img = Image.open(temp_file)\n    img = img.resize(size)\n    \n    # Remove temporary file\n    os.remove(temp_file)\n    \n    return img\n\n# Function to process a single audio file\ndef process_audio_file(args):\n    \"\"\"\n    Process a single audio file - create spectrograms from segments\n    \n    Parameters:\n    -----------\n    args : tuple\n        (bird_folder, audio_file, bird_folder_path)\n    \n    Returns:\n    --------\n    dict\n        Results statistics\n    \"\"\"\n    bird_folder, audio_file, bird_folder_path = args\n    \n    # Initialize statistics\n    stats = {\n        'specs_5sec': 0,\n        'errors': 0\n    }\n    \n    # Load audio\n    audio_path = os.path.join(bird_folder_path, audio_file)\n    try:\n        audio, sr = librosa.load(audio_path, sr=None)\n        \n        # Base filename without extension\n        base_filename = os.path.splitext(audio_file)[0]\n        \n        # Slice into 5-second segments\n        segments_5sec = segment_audio(audio, sr, 5)\n        \n        # Process 5-second segments\n        for i, segment in enumerate(segments_5sec):\n            # Create filename with index\n            segment_filename = f\"{base_filename}_{i+1:02d}\"\n            \n            # Create and save mel-spectrogram directly\n            spec_img = create_melspectrogram(segment, sr)\n            spec_img.save(f'specs_5sec/{bird_folder}/{segment_filename}.png')\n            stats['specs_5sec'] += 1\n                \n    except Exception as e:\n        print(f\"Error processing {audio_file}: {e}\")\n        stats['errors'] += 1\n    \n    return stats\n\n# Main processing function with parallel execution\ndef process_bird_audio_parallel():\n    # Create output directories\n    create_output_dirs()\n    \n    # Get list of audio folders (bird species IDs)\n    bird_folders = [f for f in os.listdir(TRAIN_AUDIO_PATH) if os.path.isdir(os.path.join(TRAIN_AUDIO_PATH, f))]\n    \n    # Initialize overall statistics\n    total_stats = {\n        'specs_5sec': 0,\n        'errors': 0\n    }\n    \n    # For each bird species folder\n    print(f\"Processing {len(bird_folders)} bird species folders...\")\n    for bird_folder in bird_folders:\n        # Create output folder for current species\n        os.makedirs(f'specs_5sec/{bird_folder}', exist_ok=True)\n        \n        # Get list of audio files\n        bird_folder_path = os.path.join(TRAIN_AUDIO_PATH, bird_folder)\n        audio_files = [f for f in os.listdir(bird_folder_path) if f.endswith(('.ogg', '.mp3', '.wav'))]\n        \n        # Prepare arguments for parallel processing\n        process_args = [(bird_folder, audio_file, bird_folder_path) for audio_file in audio_files]\n        \n        # Process files in parallel\n        print(f\"Processing {len(process_args)} files for {bird_folder}...\")\n        with concurrent.futures.ProcessPoolExecutor(max_workers=NUM_WORKERS) as executor:\n            # Submit all tasks without tqdm (to avoid ipywidgets dependency)\n            results = list(executor.map(process_audio_file, process_args))\n        \n        # Aggregate statistics\n        for result in results:\n            for key in total_stats:\n                total_stats[key] += result[key]\n    \n    return total_stats\n\n# Function to display examples of created mel-spectrograms\ndef visualize_examples():\n    # Find random examples of spectrograms\n    spec_5sec_paths = glob.glob('specs_5sec/**/*.png', recursive=True)\n    \n    # If examples found, select the first few\n    spec_5sec_examples = spec_5sec_paths[:4] if spec_5sec_paths else []\n    \n    # Create figure for display\n    n_examples = len(spec_5sec_examples)\n    if n_examples == 0:\n        print(\"No examples to display\")\n        return\n    \n    fig, axes = plt.subplots(n_examples, 1, figsize=(12, 4*n_examples))\n    \n    # If only one example, axes won't be an array\n    if n_examples == 1:\n        axes = [axes]\n    \n    # Display examples of 5-second spectrograms\n    for i, path in enumerate(spec_5sec_examples):\n        img = Image.open(path)\n        axes[i].imshow(np.array(img))\n        axes[i].set_title(f\"5-second spectrogram: {os.path.basename(path)}\")\n        axes[i].axis('off')\n    \n    plt.tight_layout()\n    plt.show()\n\n# Get statistics about processed files\ndef get_processing_stats():\n    # Count number of created files\n    specs_5sec_count = sum(len(files) for _, _, files in os.walk('specs_5sec'))\n    \n    print(f\"Processing statistics:\")\n    print(f\"- Created {specs_5sec_count} mel-spectrograms for 5-second segments\")\n\n# Run the main parallel process\nif __name__ == \"__main__\":\n    print(\"Starting parallel audio processing...\")\n    \n    # Important: matplotlib needs to be configured for non-interactive backend in parallel processing\n    plt.switch_backend('agg')\n    \n    # Process files in parallel\n    stats = process_bird_audio_parallel()\n    \n    print(\"Processing completed!\")\n    print(f\"Processing statistics from parallel processing:\")\n    print(f\"- Created {stats['specs_5sec']} mel-spectrograms for 5-second segments\")\n    print(f\"- Encountered {stats['errors']} errors during processing\")\n    \n    # Double-check with filesystem statistics\n    print(\"\\nVerifying with filesystem statistics:\")\n    get_processing_stats()\n    \n    # Display spectrogram examples\n    visualize_examples()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-11T14:26:09.079529Z","iopub.execute_input":"2025-03-11T14:26:09.079843Z","execution_failed":"2025-03-11T17:32:28.385Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}