{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\n\ndef analyze_frequency_range(audio_path, sr=32000):\n    \"\"\"Analyze frequency characteristics of an audio file\"\"\"\n    try:\n        # Load audio\n        y, _ = librosa.load(audio_path, sr=sr)\n        \n        # Compute spectrogram\n        D = librosa.stft(y)\n        S_db = librosa.amplitude_to_db(np.abs(D), ref=np.max)\n        \n        # Get frequency bins\n        freqs = librosa.fft_frequencies(sr=sr)\n        \n        # Find dominant frequencies\n        mean_spectrum = np.mean(S_db, axis=1)\n        peak_freq_idx = np.argmax(mean_spectrum)\n        peak_freq = freqs[peak_freq_idx]\n        \n        # Find frequency range containing 90% of energy\n        cumsum = np.cumsum(np.exp(mean_spectrum))\n        normalized_cumsum = cumsum / cumsum[-1]\n        lower_idx = np.where(normalized_cumsum >= 0.05)[0][0]\n        upper_idx = np.where(normalized_cumsum >= 0.95)[0][0]\n        \n        return {\n            'peak_freq': peak_freq,\n            'lower_freq': freqs[lower_idx],\n            'upper_freq': freqs[upper_idx],\n            'mean_spectrum': mean_spectrum,\n            'freqs': freqs\n        }\n    except Exception as e:\n        print(f\"Error processing {audio_path}: {e}\")\n        return None\n\ndef analyze_species_frequencies(audio_dir, species_list=None):\n    \"\"\"Analyze frequency characteristics for each species\"\"\"\n    results = {}\n    \n    # If no species list provided, analyze all species\n    if species_list is None:\n        species_list = [d.name for d in Path(audio_dir).iterdir() if d.is_dir()]\n    \n    for species in tqdm(species_list, desc=\"Analyzing species\"):\n        species_dir = Path(audio_dir) / species\n        if not species_dir.exists():\n            continue\n            \n        species_results = []\n        for audio_file in species_dir.glob(\"*.ogg\"):\n            result = analyze_frequency_range(str(audio_file))\n            if result:\n                species_results.append(result)\n        \n        if species_results:\n            # Aggregate results\n            results[species] = {\n                'peak_freq_mean': np.mean([r['peak_freq'] for r in species_results]),\n                'peak_freq_std': np.std([r['peak_freq'] for r in species_results]),\n                'lower_freq_mean': np.mean([r['lower_freq'] for r in species_results]),\n                'upper_freq_mean': np.mean([r['upper_freq'] for r in species_results]),\n                'mean_spectrum': np.mean([r['mean_spectrum'] for r in species_results], axis=0),\n                'freqs': species_results[0]['freqs']  # Same for all\n            }\n    \n    return results\n\ndef plot_species_frequency_ranges(results, output_path):\n    \"\"\"Create visualization of frequency characteristics\"\"\"\n    # Sort species by peak frequency\n    species_by_peak = sorted(results.keys(), \n                           key=lambda x: results[x]['peak_freq_mean'])\n    \n    # Create subplots\n    fig = make_subplots(rows=2, cols=1,\n                       subplot_titles=('Frequency Ranges by Species',\n                                     'Average Frequency Spectrum'))\n    \n    # Plot frequency ranges\n    fig.add_trace(\n        go.Bar(\n            name='Frequency Range',\n            x=species_by_peak,\n            y=[results[s]['upper_freq_mean'] - results[s]['lower_freq_mean'] \n               for s in species_by_peak],\n            error_y=dict(\n                type='data',\n                array=[results[s]['peak_freq_std'] for s in species_by_peak]\n            )\n        ),\n        row=1, col=1\n    )\n    \n    # Plot average spectrum for each species\n    for species in species_by_peak:\n        fig.add_trace(\n            go.Scatter(\n                name=species,\n                x=results[species]['freqs'],\n                y=results[species]['mean_spectrum'],\n                mode='lines',\n                opacity=0.6\n            ),\n            row=2, col=1\n        )\n    \n    fig.update_layout(\n        height=1000,\n        showlegend=True,\n        title_text=\"Species Frequency Analysis\"\n    )\n    \n    fig.write_html(output_path)\n\ndef main():\n    # Configuration\n    audio_dir = \"/kaggle/input/birdclef-2025/train_audio\"\n    \n    # List of poorly clustered species from previous analysis\n    problem_species = [\n        'roahaw', 'strowl1', 'linwoo1', 'blchaw1', 'savhaw1',\n        'piepuf1', 'bubcur1', 'grbhaw1', 'creoro1', 'greani1'\n    ]\n    \n    # Analyze frequencies\n    print(\"Analyzing frequency characteristics...\")\n    results = analyze_species_frequencies(audio_dir, problem_species)\n    \n    # Create visualizations\n    print(\"\\nCreating visualizations...\")\n    plot_species_frequency_ranges(results, \"/kaggle/working/frequency_analysis.html\")\n    \n    # Save detailed results\n    print(\"\\nSaving detailed results...\")\n    df_results = pd.DataFrame({\n        species: {\n            'peak_freq': results[species]['peak_freq_mean'],\n            'peak_freq_std': results[species]['peak_freq_std'],\n            'lower_freq': results[species]['lower_freq_mean'],\n            'upper_freq': results[species]['upper_freq_mean'],\n            'bandwidth': (results[species]['upper_freq_mean'] - \n                        results[species]['lower_freq_mean'])\n        }\n        for species in results\n    }).T\n    \n    df_results.to_csv(\"/kaggle/working/frequency_analysis.csv\")\n    \n    # Print recommendations\n    print(\"\\nRecommended frequency ranges for band-pass filtering:\")\n    for species in problem_species:\n        if species in results:\n            lower = results[species]['lower_freq_mean']\n            upper = results[species]['upper_freq_mean']\n            print(f\"{species}: {lower:.1f}Hz - {upper:.1f}Hz\")\n\nif __name__ == \"__main__\":\n    main() ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-30T19:11:17.878082Z","iopub.execute_input":"2025-04-30T19:11:17.878289Z","iopub.status.idle":"2025-04-30T19:15:56.961195Z","shell.execute_reply.started":"2025-04-30T19:11:17.878269Z","shell.execute_reply":"2025-04-30T19:15:56.960258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\n\ndef load_and_analyze_audio(audio_path, sr=32000):\n    \"\"\"Load audio file and compute its spectrogram\"\"\"\n    y, _ = librosa.load(audio_path, sr=sr)\n    D = librosa.stft(y)\n    S_db = librosa.amplitude_to_db(np.abs(D), ref=np.max)\n    times = librosa.times_like(S_db)\n    freqs = librosa.fft_frequencies(sr=sr)\n    return S_db, times, freqs\n\ndef create_verification_plot(audio_dir, frequency_ranges_df, output_path):\n    \"\"\"Create a comprehensive visualization to verify frequency ranges\"\"\"\n    # Get list of species from the frequency ranges DataFrame\n    species_list = frequency_ranges_df.index.tolist()\n    \n    # Create subplots - one row per species\n    fig = make_subplots(\n        rows=len(species_list), cols=1,\n        subplot_titles=[f\"{species} (Range: {frequency_ranges_df.loc[species, 'lower_freq']:.1f}Hz - {frequency_ranges_df.loc[species, 'upper_freq']:.1f}Hz)\"\n                       for species in species_list],\n        vertical_spacing=0.04\n    )\n\n    # Process each species\n    for idx, species in enumerate(species_list, 1):\n        species_dir = Path(audio_dir) / species\n        \n        # Get first audio file for the species\n        audio_files = list(species_dir.glob(\"*.ogg\"))\n        if not audio_files:\n            print(f\"Warning: No audio files found for species {species}\")\n            continue\n            \n        # Analyze first audio file\n        print(f\"Processing {species}: {audio_files[0].name}\")\n        S_db, times, freqs = load_and_analyze_audio(str(audio_files[0]))\n        \n        # Add spectrogram\n        fig.add_trace(\n            go.Heatmap(\n                z=S_db,\n                x=times,\n                y=freqs,\n                colorscale='Viridis',\n                showscale=False\n            ),\n            row=idx, col=1\n        )\n        \n        # Add horizontal lines for frequency range\n        lower_freq = frequency_ranges_df.loc[species, 'lower_freq']\n        upper_freq = frequency_ranges_df.loc[species, 'upper_freq']\n        \n        # Lower frequency line\n        fig.add_trace(\n            go.Scatter(\n                x=[times[0], times[-1]],\n                y=[lower_freq, lower_freq],\n                mode='lines',\n                line=dict(color='red', width=2),\n                name=f'Lower freq ({lower_freq:.1f}Hz)',\n                showlegend=idx == 1  # Show legend only for first species\n            ),\n            row=idx, col=1\n        )\n        \n        # Upper frequency line\n        fig.add_trace(\n            go.Scatter(\n                x=[times[0], times[-1]],\n                y=[upper_freq, upper_freq],\n                mode='lines',\n                line=dict(color='yellow', width=2),\n                name=f'Upper freq ({upper_freq:.1f}Hz)',\n                showlegend=idx == 1  # Show legend only for first species\n            ),\n            row=idx, col=1\n        )\n        \n        # Update y-axis range to focus on relevant frequencies\n        fig.update_yaxes(\n            title_text=\"Frequency (Hz)\",\n            range=[0, min(upper_freq * 1.5, 16000)],  # Cap at Nyquist frequency\n            row=idx, col=1\n        )\n        \n        # Update x-axis\n        fig.update_xaxes(\n            title_text=\"Time (s)\" if idx == len(species_list) else \"\",\n            row=idx, col=1\n        )\n\n    # Update layout\n    fig.update_layout(\n        title_text=\"Frequency Range Verification - Spectrograms with Detected Ranges\",\n        showlegend=True,\n        height=300 * len(species_list)  # Set height based on number of species\n    )\n    \n    # Save the plot\n    fig.write_html(output_path)\n    print(f\"\\nPlot saved to {output_path}\")\n\ndef main():\n    # Get the project root directory (where the frequency_analysis.csv file is)\n    project_root = \"/kaggle/working\"\n    \n    # Load the frequency analysis results\n    results_path = project_root + \"/frequency_analysis.csv\"\n    # if not results_path.exists():\n    #     print(f\"Error: Could not find frequency analysis results at {results_path}\")\n    #     return\n        \n    df = pd.read_csv(results_path, index_col=0)\n    \n    # Set paths\n    audio_dir = \"/kaggle/input/birdclef-2025/train_audio\"\n    output_path = project_root +\"/frequency_verification.html\"\n    \n    print(f\"Using audio directory: {audio_dir}\")\n    print(f\"Will save visualization to: {output_path}\")\n    \n    # Create verification plot\n    create_verification_plot(audio_dir, df, output_path)\n\nif __name__ == \"__main__\":\n    main() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T19:25:22.547814Z","iopub.execute_input":"2025-04-30T19:25:22.548318Z","iopub.status.idle":"2025-04-30T19:25:25.786661Z","shell.execute_reply.started":"2025-04-30T19:25:22.548284Z","shell.execute_reply":"2025-04-30T19:25:25.785791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}