{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install birdnetlib\n!pip install resampy ","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====================================================================\n# SCRIPT: BIRDSONG CLASSIFICATION WITH BIRDNET-ANALYZER (SUBFOLDER VERSION)\n# ====================================================================\n# This script uses the BirdNET-Analyzer library to identify bird species.\n# It processes audio files in each subfolder of a root directory and\n# creates a separate CSV file of detections for each subfolder.\n\nimport os\nimport pandas as pd\nfrom tqdm import tqdm\nfrom birdnetlib import Recording\nfrom birdnetlib.analyzer import Analyzer\n\n# --- Configuration ---\n\n# 1. Path to your ROOT folder containing subfolders of audio files.\n#    e.g., /kaggle/input/birdclef-2024/train_audio/\nROOT_AUDIO_DIR = \"/kaggle/input/birdclef-2024/train_audio/\"\n\n# 2. Path to the folder where the output CSV files will be saved.\nOUTPUT_CSV_DIR = \"/kaggle/working/birdnet_results/\"\n\n# 3. Minimum confidence threshold for a detection to be included.\nMIN_CONFIDENCE = 0.5\n\n# ====================================================================\n\ndef analyze_audio_in_subfolders(root_dir, output_dir):\n    \"\"\"\n    Analyzes audio files in each subfolder of a root directory and saves\n    a separate CSV for each.\n    \"\"\"\n    print(\"--- Birdsong Analysis with BirdNET (Subfolder Mode) ---\")\n\n    # Step 1: Initialize the BirdNET Analyzer.\n    try:\n        print(\"Loading BirdNET Analyzer model...\")\n        analyzer = Analyzer()\n    except Exception as e:\n        print(f\"❌ ERROR: Failed to load the BirdNET Analyzer. Reason: {e}\")\n        return\n\n    # Step 2: Create the main output directory if it doesn't exist.\n    os.makedirs(output_dir, exist_ok=True)\n    print(f\"Output CSVs will be saved to: '{output_dir}'\")\n\n    # Step 3: Find all subdirectories in the root directory.\n    subfolders = [f.path for f in os.scandir(root_dir) if f.is_dir()]\n\n    if not subfolders:\n        print(f\"⚠️ Warning: No subfolders found in '{root_dir}'.\")\n        return\n\n    print(f\"Found {len(subfolders)} subfolders to process.\")\n\n    # Step 4: Process each subfolder.\n    for folder_path in tqdm(subfolders, desc=\"Processing Folders\"):\n        folder_name = os.path.basename(folder_path)\n        tqdm.write(f\"\\n--- Processing folder: {folder_name} ---\")\n\n        audio_files = [os.path.join(folder_path, f) for f in os.listdir(folder_path)\n                       if f.endswith((\".ogg\", \".mp3\", \".wav\"))]\n\n        if not audio_files:\n            tqdm.write(f\"No audio files found in '{folder_name}', skipping.\")\n            continue\n\n        all_results = []\n        for audio_path in tqdm(audio_files, desc=f\"Analyzing {folder_name}\", leave=False):\n            try:\n                recording = Recording(\n                    analyzer=analyzer,\n                    path=audio_path,\n                    min_conf=MIN_CONFIDENCE,\n                )\n                recording.analyze()\n\n                if recording.detections:\n                    filename = os.path.basename(audio_path)\n                    for d in recording.detections:\n                        all_results.append({\n                            \"filename\": filename,\n                            \"start_time\": d[\"start_time\"],\n                            \"end_time\": d[\"end_time\"],\n                            \"common_name\": d[\"common_name\"],\n                            \"scientific_name\": d[\"scientific_name\"],\n                            \"confidence\": d[\"confidence\"],\n                        })\n\n            except Exception as e:\n                tqdm.write(f\"❌ ERROR: Could not process file {audio_path}. Reason: {e}\")\n\n        # Step 5: Save a separate CSV for this subfolder's results.\n        if not all_results:\n            tqdm.write(f\"No detections found for folder '{folder_name}'.\")\n            continue\n\n        results_df = pd.DataFrame(all_results)\n        results_df.sort_values(by=[\"filename\", \"start_time\"], inplace=True)\n        \n        # Name the CSV file after the subfolder.\n        output_csv_path = os.path.join(output_dir, f\"{folder_name}.csv\")\n        results_df.to_csv(output_csv_path, index=False)\n        tqdm.write(f\"✅ Results for '{folder_name}' saved to '{output_csv_path}'\")\n\n    print(\"\\n--- All Folders Processed ---\")\n\n\n# --- Execution Block ---\nif __name__ == \"__main__\":\n    if not os.path.exists(ROOT_AUDIO_DIR):\n        print(f\"❌ ERROR: Root audio directory not found at '{ROOT_AUDIO_DIR}'.\")\n    else:\n        analyze_audio_in_subfolders(\n            root_dir=ROOT_AUDIO_DIR,\n            output_dir=OUTPUT_CSV_DIR\n        )","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null}]}