{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### What is it?\n\nThis code will create 10 second audio chunks and 224x224 PNG melspectrograms for BirdCLEF+ 2025. *Runtime about 5 hours.\n- Please find output of this code at the [BirdCLEF+ 2025, MelSPECs (10 sec) Dataset](https://www.kaggle.com/datasets/samvelkoch/birdclef-2025-melspecs-10-sec/)\n- Addon: 5 seconds Melspectrograms [Code](https://www.kaggle.com/code/samvelkoch/birdclef2025-dataprep-specs-5-sec/) and [Dataset](https://www.kaggle.com/datasets/samvelkoch/birdclef-2025-melspecs-5-sec)\n\nHave fun... Thank you 🙏","metadata":{}},{"cell_type":"code","source":"# Audio Processing for BirdCLEF 2025\n# Slicing audio into 10-second segments and creating mel-spectrograms with parallel processing\n\nimport os\nimport numpy as np\nimport librosa\nimport librosa.display\nimport matplotlib.pyplot as plt\nfrom matplotlib import cm\nimport soundfile as sf\nimport warnings\nwarnings.filterwarnings('ignore')\nfrom PIL import Image\nimport glob\nimport concurrent.futures\nimport multiprocessing\nimport tqdm\ntqdm.tqdm = tqdm.tqdm_notebook = tqdm.tqdm\n\n# Path configuration\nBASE_PATH = '/kaggle/input/birdclef-2025'\nTRAIN_AUDIO_PATH = os.path.join(BASE_PATH, 'train_audio')\n\n# Number of workers for parallel processing\n# Using 80% of available CPUs is usually a good balance\nNUM_WORKERS = max(1, int(multiprocessing.cpu_count() * 0.8))\nprint(f\"Using {NUM_WORKERS} workers for parallel processing\")\n\n# Create destination directories\ndef create_output_dirs():\n    # For 10-second audio\n    os.makedirs('audio_10sec', exist_ok=True)\n    os.makedirs('specs_10sec', exist_ok=True)\n    \n    print(\"Output directories created\")\n\n# Function to slice audio into segments\ndef segment_audio(audio, sr, duration):\n    \"\"\"\n    Slices audio into segments of specified duration\n    \n    Parameters:\n    -----------\n    audio : np.array\n        Audio data\n    sr : int\n        Sample rate\n    duration : int\n        Segment duration in seconds\n    \n    Returns:\n    --------\n    list\n        List of audio segments\n    \"\"\"\n    # Calculate number of samples for the desired duration\n    samples_per_segment = int(sr * duration)\n    \n    # Calculate number of full segments\n    num_segments = len(audio) // samples_per_segment\n    \n    segments = []\n    for i in range(num_segments):\n        start = i * samples_per_segment\n        end = start + samples_per_segment\n        segment = audio[start:end]\n        segments.append(segment)\n    \n    return segments\n\n# Function to create mel-spectrogram\ndef create_melspectrogram(audio, sr, size=(224, 224)):\n    \"\"\"\n    Creates a mel-spectrogram from audio data\n    \n    Parameters:\n    -----------\n    audio : np.array\n        Audio data\n    sr : int\n        Sample rate\n    size : tuple\n        Output image size (width, height)\n    \n    Returns:\n    --------\n    PIL.Image\n        Mel-spectrogram image of size (width, height)\n    \"\"\"\n    # Calculate mel-spectrogram\n    melspec = librosa.feature.melspectrogram(\n        y=audio, \n        sr=sr, \n        n_mels=128,\n        fmax=sr/2\n    )\n    \n    # Convert to decibels\n    melspec_db = librosa.power_to_db(melspec, ref=np.max)\n    \n    # Create figure of required size\n    fig, ax = plt.subplots(figsize=(10, 4))\n    \n    # Display spectrogram\n    img = librosa.display.specshow(\n        melspec_db, \n        x_axis='time', \n        y_axis='mel', \n        sr=sr,\n        cmap=cm.viridis,\n        ax=ax\n    )\n    \n    # Remove axes and padding\n    plt.axis('off')\n    plt.tight_layout(pad=0)\n    \n    # Use a unique temporary filename for parallel processing\n    import uuid\n    temp_file = f'temp_spec_{uuid.uuid4()}.png'\n    plt.savefig(temp_file, bbox_inches='tight', pad_inches=0)\n    plt.close(fig)\n    \n    # Open and resize the image\n    img = Image.open(temp_file)\n    img = img.resize(size)\n    \n    # Remove temporary file\n    os.remove(temp_file)\n    \n    return img\n\n# Function to process a single audio file\ndef process_audio_file(args):\n    \"\"\"\n    Process a single audio file - create segments and spectrograms\n    \n    Parameters:\n    -----------\n    args : tuple\n        (bird_folder, audio_file, bird_folder_path)\n    \n    Returns:\n    --------\n    dict\n        Results statistics\n    \"\"\"\n    bird_folder, audio_file, bird_folder_path = args\n    \n    # Initialize statistics\n    stats = {\n        'audio_10sec': 0,\n        'specs_10sec': 0,\n        'errors': 0\n    }\n    \n    # Load audio\n    audio_path = os.path.join(bird_folder_path, audio_file)\n    try:\n        audio, sr = librosa.load(audio_path, sr=None)\n        \n        # Base filename without extension\n        base_filename = os.path.splitext(audio_file)[0]\n        \n        # Slice into 10-second segments\n        segments_10sec = segment_audio(audio, sr, 10)\n        \n        # Process 10-second segments\n        for i, segment in enumerate(segments_10sec):\n            # Create filename with index\n            segment_filename = f\"{base_filename}_{i+1:02d}.ogg\"\n            \n            # Save audio segment\n            sf.write(f'audio_10sec/{bird_folder}/{segment_filename}', segment, sr)\n            stats['audio_10sec'] += 1\n            \n            # Create and save mel-spectrogram\n            spec_img = create_melspectrogram(segment, sr)\n            spec_img.save(f'specs_10sec/{bird_folder}/{os.path.splitext(segment_filename)[0]}.png')\n            stats['specs_10sec'] += 1\n                \n    except Exception as e:\n        print(f\"Error processing {audio_file}: {e}\")\n        stats['errors'] += 1\n    \n    return stats\n\n# Main processing function with parallel execution\ndef process_bird_audio_parallel():\n    # Create output directories\n    create_output_dirs()\n    \n    # Get list of audio folders (bird species IDs)\n    bird_folders = [f for f in os.listdir(TRAIN_AUDIO_PATH) if os.path.isdir(os.path.join(TRAIN_AUDIO_PATH, f))]\n    \n    # Initialize overall statistics\n    total_stats = {\n        'audio_10sec': 0,\n        'specs_10sec': 0,\n        'errors': 0\n    }\n    \n    # For each bird species folder\n    print(f\"Processing {len(bird_folders)} bird species folders...\")\n    for bird_folder in bird_folders:\n        # Create output folders for current species\n        os.makedirs(f'audio_10sec/{bird_folder}', exist_ok=True)\n        os.makedirs(f'specs_10sec/{bird_folder}', exist_ok=True)\n        \n        # Get list of audio files\n        bird_folder_path = os.path.join(TRAIN_AUDIO_PATH, bird_folder)\n        audio_files = [f for f in os.listdir(bird_folder_path) if f.endswith(('.ogg', '.mp3', '.wav'))]\n        \n        # Prepare arguments for parallel processing\n        process_args = [(bird_folder, audio_file, bird_folder_path) for audio_file in audio_files]\n        \n        # Process files in parallel\n        print(f\"Processing {len(process_args)} files for {bird_folder}...\")\n        with concurrent.futures.ProcessPoolExecutor(max_workers=NUM_WORKERS) as executor:\n            # Submit all tasks without tqdm (to avoid ipywidgets dependency)\n            results = list(executor.map(process_audio_file, process_args))\n        \n        # Aggregate statistics\n        for result in results:\n            for key in total_stats:\n                total_stats[key] += result[key]\n    \n    return total_stats\n\n# Function to display examples of created mel-spectrograms\ndef visualize_examples():\n    # Find random examples of spectrograms\n    spec_10sec_paths = glob.glob('specs_10sec/**/*.png', recursive=True)\n    \n    # If examples found, select the first few\n    spec_10sec_examples = spec_10sec_paths[:4] if spec_10sec_paths else []\n    \n    # Create figure for display\n    n_examples = len(spec_10sec_examples)\n    if n_examples == 0:\n        print(\"No examples to display\")\n        return\n    \n    fig, axes = plt.subplots(n_examples, 1, figsize=(12, 4*n_examples))\n    \n    # If only one example, axes won't be an array\n    if n_examples == 1:\n        axes = [axes]\n    \n    # Display examples of 10-second spectrograms\n    for i, path in enumerate(spec_10sec_examples):\n        img = Image.open(path)\n        axes[i].imshow(np.array(img))\n        axes[i].set_title(f\"10-second spectrogram: {os.path.basename(path)}\")\n        axes[i].axis('off')\n    \n    plt.tight_layout()\n    plt.show()\n\n# Get statistics about processed files\ndef get_processing_stats():\n    # Count number of created files\n    audio_10sec_count = sum(len(files) for _, _, files in os.walk('audio_10sec'))\n    specs_10sec_count = sum(len(files) for _, _, files in os.walk('specs_10sec'))\n    \n    print(f\"Processing statistics:\")\n    print(f\"- Created {audio_10sec_count} 10-second audio segments\")\n    print(f\"- Created {specs_10sec_count} mel-spectrograms for 10-second segments\")\n\n# Run the main parallel process\nif __name__ == \"__main__\":\n    print(\"Starting parallel audio processing...\")\n    \n    # Important: matplotlib needs to be configured for non-interactive backend in parallel processing\n    plt.switch_backend('agg')\n    \n    # Process files in parallel\n    stats = process_bird_audio_parallel()\n    \n    print(\"Processing completed!\")\n    print(f\"Processing statistics from parallel processing:\")\n    print(f\"- Created {stats['audio_10sec']} 10-second audio segments\")\n    print(f\"- Created {stats['specs_10sec']} mel-spectrograms for 10-second segments\")\n    print(f\"- Encountered {stats['errors']} errors during processing\")\n    \n    # Double-check with filesystem statistics\n    print(\"\\nVerifying with filesystem statistics:\")\n    get_processing_stats()\n    \n    # Display spectrogram examples\n    visualize_examples()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-11T14:26:09.079529Z","iopub.execute_input":"2025-03-11T14:26:09.079843Z","execution_failed":"2025-03-11T17:32:28.385Z"}},"outputs":[],"execution_count":null}]}