{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30683,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Deep Learning-Enhanced Real-Time Audio Spectrum Analyzer for Bioacoustics\nIn this notebook, we explore a hybrid digital signal processing and deep learning pipeline for real-time bioacoustic signal enhancement and spectrum analysis, focused on the **BirdCLEF 2024** dataset.\n\n### Key Objectives:\n1. **Adaptive Preprocessing**: Introduce a novel **Adaptive Spectral Gating Preprocessing (ASGP)** pipeline to dynamically estimate environmental noise floors and suppress out-of-band interference.\n2. **Deep Learning Denoising**: Deploy a pretrained **Dual-Signal Transformation LSTM Network (DTLN)** to estimate spectral masks and perform real-time speech/audio denoising.\n3. **Advanced Visualizations**: Compare audio representations across multiple domains (time waveforms, linear spectrograms, Mel-spectrograms, spectral centroids, and chromagrams).\n4. **Metadata & Distribution Analysis**: Analyze the statistical correlation between human-labeled ratings and model-derived signal-to-noise ratios (SNR), and compare SNR distributions across training and unlabeled soundscapes (test) data.\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"# use this repo's denoising model, star it maybe if you think useful\n# may need update models for bird sounds enhancement \n!git clone https://github.com/lhwcv/DTLN_pytorch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-21T11:58:19.326929Z","iopub.execute_input":"2026-06-21T11:58:19.327856Z","iopub.status.idle":"2026-06-21T11:58:22.189062Z","shell.execute_reply.started":"2026-06-21T11:58:19.327828Z","shell.execute_reply":"2026-06-21T11:58:22.188140Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\nimport random\nimport torch\nimport sys\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport scipy.signal\n\n# Set styling\nplt.style.use('ggplot')\nsns.set_theme(style=\"whitegrid\")\n\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\n\nsys.path.append('/kaggle/working/DTLN_pytorch/')\nfrom DTLN_model import Pytorch_DTLN\nfrom audio_io import wav_read\n","metadata":{"execution":{"iopub.status.busy":"2026-06-21T11:58:22.190998Z","iopub.execute_input":"2026-06-21T11:58:22.191348Z","iopub.status.idle":"2026-06-21T11:58:25.940133Z","shell.execute_reply.started":"2026-06-21T11:58:22.191320Z","shell.execute_reply":"2026-06-21T11:58:25.938979Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Pytorch_DTLN().cuda()\nprint('==> load model.. ', )\nmodel.load_state_dict(torch.load('/kaggle/working/DTLN_pytorch/pretrained/model.pth'))\nmodel = model.eval()","metadata":{"execution":{"iopub.status.busy":"2026-06-21T11:58:25.941681Z","iopub.execute_input":"2026-06-21T11:58:25.942029Z","iopub.status.idle":"2026-06-21T11:58:26.275266Z","shell.execute_reply.started":"2026-06-21T11:58:25.942001Z","shell.execute_reply":"2026-06-21T11:58:26.274229Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def calculate_snr_by_model(filename):\n    def snr(clean, noisy):\n        # Align lengths to prevent broadcasting errors\n        min_len = min(len(clean), len(noisy))\n        clean_aligned = clean[:min_len]\n        noisy_aligned = noisy[:min_len]\n        noise = noisy_aligned - clean_aligned\n        denom = np.sum(noise**2)\n        if denom == 0:\n            return 100.0\n        return 10 * np.log10(np.sum(clean_aligned**2) / denom)\n\n    x, fs = wav_read(filename, tgt_fs=16000)\n    xt = torch.from_numpy(x).unsqueeze(0).cuda()\n    \n    with torch.no_grad():\n        clean = model(xt).squeeze().cpu().numpy()\n    \n    snr_value = snr(clean, x)\n    \n    return snr_value, clean, fs\n","metadata":{"execution":{"iopub.status.busy":"2026-06-21T11:58:26.276452Z","iopub.execute_input":"2026-06-21T11:58:26.276725Z","iopub.status.idle":"2026-06-21T11:58:26.283892Z","shell.execute_reply.started":"2026-06-21T11:58:26.276704Z","shell.execute_reply":"2026-06-21T11:58:26.283049Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def adaptive_spectral_gating_preprocessing(y, sr, noise_reduction_factor=0.85):\n    \"\"\"\n    Adaptive Spectral Gating Preprocessing (ASGP) for bioacoustic recordings.\n    1. Butterworth Bandpass Filter (1000Hz - 7500Hz) to focus on bird vocalizations\n       and remove low-frequency (wind, traffic) and high-frequency (static, sensor) noise.\n    2. Spectral Subtraction based on the 15th percentile noise floor estimation.\n    \"\"\"\n    # 1. Bandpass Filter\n    lowcut = 1000.0\n    highcut = 7500.0\n    nyq = 0.5 * sr\n    low = lowcut / nyq\n    high = highcut / nyq\n    \n    # 4th-order Butterworth bandpass\n    b, a = scipy.signal.butter(4, [low, high], btype='band')\n    y_filtered = scipy.signal.filtfilt(b, a, y)\n    \n    # 2. Spectral Subtraction (Noise Floor Estimation)\n    stft = librosa.stft(y_filtered, n_fft=512, hop_length=128)\n    magnitude, phase = librosa.magphase(stft)\n    \n    # Estimate background noise floor using the 15th percentile along time axis\n    noise_floor = np.percentile(magnitude, 15, axis=1, keepdims=True)\n    \n    # Perform spectral gating/subtraction\n    magnitude_denoised = magnitude - noise_reduction_factor * noise_floor\n    magnitude_denoised = np.maximum(magnitude_denoised, 1e-5) # Avoid absolute silence/numerical issues\n    \n    # Reconstruct spectrogram and convert back to time domain\n    stft_denoised = magnitude_denoised * phase\n    y_preprocessed = librosa.istft(stft_denoised, hop_length=128)\n    \n    # Normalize amplitude to [-1, 1]\n    if np.max(np.abs(y_preprocessed)) > 0:\n        y_preprocessed = y_preprocessed / np.max(np.abs(y_preprocessed))\n        \n    return y_preprocessed\n\ndef calculate_snr_with_preprocessing(filename):\n    \"\"\"\n    Computes SNR before and after ASGP preprocessing, returning both waveforms and clean signal.\n    \"\"\"\n    def snr(clean, noisy):\n        # Align lengths to prevent broadcasting errors\n        min_len = min(len(clean), len(noisy))\n        clean_aligned = clean[:min_len]\n        noisy_aligned = noisy[:min_len]\n        noise = noisy_aligned - clean_aligned\n        # Avoid division by zero\n        denom = np.sum(noise**2)\n        if denom == 0:\n            return 100.0\n        return 10 * np.log10(np.sum(clean_aligned**2) / denom)\n\n    # Read audio (DTLN expects 16kHz)\n    x, fs = wav_read(filename, tgt_fs=16000)\n    \n    # 1. Apply ASGP Preprocessing\n    x_prep = adaptive_spectral_gating_preprocessing(x, fs)\n    \n    # 2. Feed preprocessed audio to DTLN model\n    xt = torch.from_numpy(x_prep).float().unsqueeze(0).cuda()\n    with torch.no_grad():\n        clean = model(xt).squeeze().cpu().numpy()\n        \n    # Calculate SNRs\n    snr_raw = snr(clean, x)\n    snr_prep = snr(clean, x_prep)\n    \n    return snr_raw, snr_prep, x_prep, clean, fs\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-21T11:58:26.287122Z","iopub.execute_input":"2026-06-21T11:58:26.287594Z","iopub.status.idle":"2026-06-21T11:58:26.298827Z","shell.execute_reply.started":"2026-06-21T11:58:26.287570Z","shell.execute_reply":"2026-06-21T11:58:26.298023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_audio_comprehensive(filename):\n    \"\"\"\n    Plots a multi-resolution feature analysis comparing Raw, Preprocessed (ASGP), and DTLN-Denoised audio.\n    Visualizes waveforms, linear spectrograms, Mel-spectrograms, spectral centroids, and chromagrams.\n    \"\"\"\n    # Compute representations and SNRs\n    snr_raw, snr_prep, y_prep, y_clean, sr = calculate_snr_with_preprocessing(filename)\n    \n    # Load raw audio at 16kHz for uniform comparison\n    y_raw, _ = librosa.load(filename, sr=16000)\n    \n    # Align lengths\n    min_len = min(len(y_raw), len(y_prep), len(y_clean))\n    y_raw = y_raw[:min_len]\n    y_prep = y_prep[:min_len]\n    y_clean = y_clean[:min_len]\n    \n    time_axis = np.arange(min_len) / sr\n    \n    # Display audio players\n    print(\"--- AUDIO PLAYBACK ---\")\n    print(\"Raw Input Audio:\")\n    ipd.display(ipd.Audio(y_raw, rate=sr))\n    print(\"ASGP Preprocessed Audio:\")\n    ipd.display(ipd.Audio(y_prep, rate=sr))\n    print(\"DTLN Denoised Audio:\")\n    ipd.display(ipd.Audio(y_clean, rate=sr))\n    \n    # Plotting Grid\n    fig = plt.figure(figsize=(22, 25))\n    gs = fig.add_gridspec(5, 3, height_ratios=[1, 1.2, 1.2, 1, 1.2])\n    \n    # Row 1: Waveforms\n    ax_w1 = fig.add_subplot(gs[0, 0])\n    ax_w1.plot(time_axis, y_raw, color='crimson', alpha=0.7)\n    ax_w1.set_title(f'Raw Time-Domain Waveform\\n(SNR: {snr_raw:.2f} dB)')\n    ax_w1.set_ylabel('Amplitude')\n    ax_w1.set_xlabel('Time (s)')\n    \n    ax_w2 = fig.add_subplot(gs[0, 1])\n    ax_w2.plot(time_axis, y_prep, color='dodgerblue', alpha=0.7)\n    ax_w2.set_title(f'ASGP Preprocessed Waveform\\n(SNR: {snr_prep:.2f} dB)')\n    ax_w2.set_xlabel('Time (s)')\n    \n    ax_w3 = fig.add_subplot(gs[0, 2])\n    ax_w3.plot(time_axis, y_clean, color='forestgreen', alpha=0.7)\n    ax_w3.set_title(f'DTLN Denoised Waveform\\n(SNR Improvement: {snr_prep - snr_raw:.2f} dB)')\n    ax_w3.set_xlabel('Time (s)')\n    \n    # Row 2: Linear Spectrograms\n    def plot_spec(y, ax, title):\n        stft = librosa.amplitude_to_db(np.abs(librosa.stft(y, n_fft=512, hop_length=128)), ref=np.max)\n        img = librosa.display.specshow(stft, sr=sr, hop_length=128, x_axis='time', y_axis='linear', ax=ax, cmap='magma')\n        ax.set_title(title)\n        ax.set_ylim(0, 8000) # focus on audible range\n        return img\n\n    ax_s1 = fig.add_subplot(gs[1, 0])\n    plot_spec(y_raw, ax_s1, 'Raw Linear Spectrogram')\n    \n    ax_s2 = fig.add_subplot(gs[1, 1])\n    plot_spec(y_prep, ax_s2, 'ASGP Preprocessed Spectrogram')\n    \n    ax_s3 = fig.add_subplot(gs[1, 2])\n    plot_spec(y_clean, ax_s3, 'DTLN Denoised Spectrogram')\n    \n    # Row 3: Mel-Spectrograms\n    def plot_mel(y, ax, title):\n        melspec = librosa.power_to_db(librosa.feature.melspectrogram(y=y, sr=sr, n_fft=512, hop_length=128, n_mels=128), ref=np.max)\n        img = librosa.display.specshow(melspec, sr=sr, hop_length=128, x_axis='time', y_axis='mel', ax=ax, cmap='viridis')\n        ax.set_title(title)\n        return img\n\n    ax_m1 = fig.add_subplot(gs[2, 0])\n    plot_mel(y_raw, ax_m1, 'Raw Mel Spectrogram')\n    \n    ax_m2 = fig.add_subplot(gs[2, 1])\n    plot_mel(y_prep, ax_m2, 'ASGP Preprocessed Mel Spectrogram')\n    \n    ax_m3 = fig.add_subplot(gs[2, 2])\n    plot_mel(y_clean, ax_m3, 'DTLN Denoised Mel Spectrogram')\n    \n    # Row 4: Spectral Features (Spectral Centroid & Zero Crossing Rate)\n    ax_f = fig.add_subplot(gs[3, :])\n    sc_raw = librosa.feature.spectral_centroid(y=y_raw, sr=sr, n_fft=512, hop_length=128)[0]\n    sc_prep = librosa.feature.spectral_centroid(y=y_prep, sr=sr, n_fft=512, hop_length=128)[0]\n    sc_clean = librosa.feature.spectral_centroid(y=y_clean, sr=sr, n_fft=512, hop_length=128)[0]\n    \n    frames = range(len(sc_raw))\n    t_frames = librosa.frames_to_time(frames, sr=sr, hop_length=128)\n    \n    ax_f.plot(t_frames, sc_raw, label='Raw Input', color='crimson', alpha=0.5, linestyle='--')\n    ax_f.plot(t_frames, sc_prep, label='ASGP Preprocessed', color='dodgerblue', alpha=0.7)\n    ax_f.plot(t_frames, sc_clean, label='DTLN Denoised', color='forestgreen', alpha=0.8, linewidth=2)\n    ax_f.set_title('Spectral Centroid Tracking (Out-of-band Noise Suppression)')\n    ax_f.set_xlabel('Time (s)')\n    ax_f.set_ylabel('Centroid Frequency (Hz)')\n    ax_f.legend(loc='upper right')\n    \n    # Row 5: Chromagrams (Pitch/Harmonic Preservation)\n    def plot_chroma(y, ax, title):\n        chroma = librosa.feature.chroma_stft(y=y, sr=sr, n_fft=512, hop_length=128)\n        img = librosa.display.specshow(chroma, y_axis='chroma', x_axis='time', ax=ax, cmap='coolwarm')\n        ax.set_title(title)\n        return img\n\n    ax_c1 = fig.add_subplot(gs[4, 0])\n    plot_chroma(y_raw, ax_c1, 'Raw Chromagram')\n    \n    ax_c2 = fig.add_subplot(gs[4, 1])\n    plot_chroma(y_prep, ax_c2, 'ASGP Chromagram')\n    \n    ax_c3 = fig.add_subplot(gs[4, 2])\n    plot_chroma(y_clean, ax_c3, 'DTLN Chromagram')\n    \n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2026-06-21T11:58:26.300034Z","iopub.execute_input":"2026-06-21T11:58:26.300638Z","iopub.status.idle":"2026-06-21T11:58:26.317709Z","shell.execute_reply.started":"2026-06-21T11:58:26.300616Z","shell.execute_reply":"2026-06-21T11:58:26.316864Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"filepath = '/kaggle/input/birdclef-2024/train_audio/asbfly/XC134896.ogg'\n","metadata":{"execution":{"iopub.status.busy":"2026-06-21T11:58:26.318799Z","iopub.execute_input":"2026-06-21T11:58:26.319354Z","iopub.status.idle":"2026-06-21T11:58:26.328193Z","shell.execute_reply.started":"2026-06-21T11:58:26.319333Z","shell.execute_reply":"2026-06-21T11:58:26.327284Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"visualize_audio_comprehensive(filepath)\n","metadata":{"execution":{"iopub.status.busy":"2026-06-21T11:58:26.329352Z","iopub.execute_input":"2026-06-21T11:58:26.329616Z","iopub.status.idle":"2026-06-21T11:58:44.066906Z","shell.execute_reply.started":"2026-06-21T11:58:26.329596Z","shell.execute_reply":"2026-06-21T11:58:44.063316Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"snr_raw, snr_prep, y_prep, y_clean, sample_rate = calculate_snr_with_preprocessing(filepath)\nprint(f'Raw Audio Model-estimated SNR: {snr_raw:.2f} dB')\nprint(f'ASGP Preprocessed Model-estimated SNR: {snr_prep:.2f} dB')\nprint(f'SNR Improvement: {snr_prep - snr_raw:.2f} dB')\n","metadata":{"execution":{"iopub.status.busy":"2026-06-21T11:58:44.068069Z","iopub.execute_input":"2026-06-21T11:58:44.068632Z","iopub.status.idle":"2026-06-21T11:58:44.347078Z","shell.execute_reply.started":"2026-06-21T11:58:44.068602Z","shell.execute_reply":"2026-06-21T11:58:44.345994Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Comprehensive visualizations displayed above.\n","metadata":{"execution":{"iopub.status.busy":"2026-06-21T11:58:44.348510Z","iopub.execute_input":"2026-06-21T11:58:44.348998Z","iopub.status.idle":"2026-06-21T11:58:44.353406Z","shell.execute_reply.started":"2026-06-21T11:58:44.348964Z","shell.execute_reply":"2026-06-21T11:58:44.352289Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\ndf['filename'] = '/kaggle/input/birdclef-2024/train_audio/' + df['filename']","metadata":{"execution":{"iopub.status.busy":"2026-06-21T11:58:44.354591Z","iopub.execute_input":"2026-06-21T11:58:44.355357Z","iopub.status.idle":"2026-06-21T11:58:44.501732Z","shell.execute_reply.started":"2026-06-21T11:58:44.355210Z","shell.execute_reply":"2026-06-21T11:58:44.500903Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"part_df = df.sample(1000, random_state=42).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2026-06-21T11:58:44.502940Z","iopub.execute_input":"2026-06-21T11:58:44.503664Z","iopub.status.idle":"2026-06-21T11:58:44.515706Z","shell.execute_reply.started":"2026-06-21T11:58:44.503631Z","shell.execute_reply":"2026-06-21T11:58:44.514632Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tqdm\nsnrs_raw = []\nsnrs_prep = []\nfilepaths = part_df['filename'].tolist()\n\nfor filepath in tqdm.tqdm(filepaths):\n    try:\n        s_raw, s_prep, _, _, _ = calculate_snr_with_preprocessing(filepath)\n        snrs_raw.append(s_raw)\n        snrs_prep.append(s_prep)\n    except Exception as e:\n        snrs_raw.append(np.nan)\n        snrs_prep.append(np.nan)\n\npart_df['snr_raw'] = snrs_raw\npart_df['snr_prep'] = snrs_prep\npart_df['snr'] = snrs_raw  # Keep for backward compatibility\n","metadata":{"execution":{"iopub.status.busy":"2026-06-21T11:58:44.516928Z","iopub.execute_input":"2026-06-21T11:58:44.517284Z","iopub.status.idle":"2026-06-21T12:04:59.460648Z","shell.execute_reply.started":"2026-06-21T11:58:44.517254Z","shell.execute_reply":"2026-06-21T12:04:59.459683Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot 1: Violin & Box plots of SNR grouped by Rating\nfig, axs = plt.subplots(nrows=2, ncols=1, figsize=(14, 12), sharex=True)\n\nsns.violinplot(data=part_df, x='rating', y='snr_prep', ax=axs[0], palette='Blues')\naxs[0].set_title('Violin Plot: Preprocessed SNR (ASGP) Distribution by Human Rating')\naxs[0].set_ylabel('SNR (dB)')\naxs[0].set_xlabel('')\n\nsns.boxplot(data=part_df, x='rating', y='snr_prep', ax=axs[1], palette='Oranges')\naxs[1].set_title('Box Plot: Preprocessed SNR (ASGP) Distribution by Human Rating')\naxs[1].set_ylabel('SNR (dB)')\naxs[1].set_xlabel('Metadata Rating')\n\nplt.tight_layout()\nplt.show()\n\n# Plot 2: Joint KDE and Scatter plot of SNR vs Rating\ng = sns.jointplot(\n    data=part_df.dropna(subset=['snr_prep', 'rating']),\n    x='rating', y='snr_prep',\n    kind='reg', height=8,\n    scatter_kws={'alpha': 0.3, 'color': 'darkslateblue'},\n    line_kws={'color': 'red', 'linewidth': 2}\n)\ng.fig.suptitle('Joint Distribution & Regression: Rating vs. Preprocessed SNR', y=1.02)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-21T12:04:59.463874Z","iopub.execute_input":"2026-06-21T12:04:59.464174Z","iopub.status.idle":"2026-06-21T12:05:01.074116Z","shell.execute_reply.started":"2026-06-21T12:04:59.464152Z","shell.execute_reply":"2026-06-21T12:05:01.073169Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# let's look rating > 4 but preprocessed snr < -30\ncheck_df = part_df[part_df['rating'] > 4]\ncheck_df = check_df[check_df['snr_prep'] < -30].dropna()\nprint(f\"Number of files with rating > 4 but SNR < -30 dB: {len(check_df)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-21T12:05:01.075306Z","iopub.execute_input":"2026-06-21T12:05:01.075593Z","iopub.status.idle":"2026-06-21T12:05:01.083507Z","shell.execute_reply.started":"2026-06-21T12:05:01.075572Z","shell.execute_reply":"2026-06-21T12:05:01.082628Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if len(check_df) > 0:\n    visualize_audio_comprehensive(check_df['filename'].iloc[0])\nelse:\n    print(\"No files match the search criteria.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-21T12:05:01.084525Z","iopub.execute_input":"2026-06-21T12:05:01.084738Z","iopub.status.idle":"2026-06-21T12:05:08.641487Z","shell.execute_reply.started":"2026-06-21T12:05:01.084720Z","shell.execute_reply":"2026-06-21T12:05:08.640450Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### It seems sound good, we should develop a denoise model for bird sound","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Plot 3: Overlapping Histogram and KDE for Train Data (Raw vs Preprocessed SNR)\nplt.figure(figsize=(12, 6))\nsns.histplot(part_df['snr_raw'], color='crimson', label='Raw SNR', kde=True, bins=40, alpha=0.4, stat='density')\nsns.histplot(part_df['snr_prep'], color='dodgerblue', label='Preprocessed SNR (ASGP)', kde=True, bins=40, alpha=0.5, stat='density')\nplt.title('Train Data SNR Distribution: Raw vs. ASGP Preprocessed')\nplt.xlabel('Estimated SNR (dB)')\nplt.ylabel('Density')\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-21T12:05:08.642706Z","iopub.execute_input":"2026-06-21T12:05:08.643029Z","iopub.status.idle":"2026-06-21T12:05:09.190732Z","shell.execute_reply.started":"2026-06-21T12:05:08.643003Z","shell.execute_reply":"2026-06-21T12:05:09.189815Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## let's look unlabeled_soundscapes SNR distribution\n","metadata":{}},{"cell_type":"code","source":"test_files = glob.glob('/kaggle/input/birdclef-2024/unlabeled_soundscapes/*.ogg')\npart_test_files = random.sample(test_files, min(len(test_files), 500))\ntest_snrs_raw = []\ntest_snrs_prep = []\n\nfor filepath in tqdm.tqdm(part_test_files):\n    try:\n        s_raw, s_prep, _, _, _ = calculate_snr_with_preprocessing(filepath)\n        test_snrs_raw.append(s_raw)\n        test_snrs_prep.append(s_prep)\n    except Exception as e:\n        test_snrs_raw.append(np.nan)\n        test_snrs_prep.append(np.nan)\n\ntest_snrs_raw = [x for x in test_snrs_raw if not np.isnan(x)]\ntest_snrs_prep = [x for x in test_snrs_prep if not np.isnan(x)]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-21T12:05:09.191822Z","iopub.execute_input":"2026-06-21T12:05:09.192169Z","iopub.status.idle":"2026-06-21T12:21:19.928042Z","shell.execute_reply.started":"2026-06-21T12:05:09.192146Z","shell.execute_reply":"2026-06-21T12:21:19.926975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot 4: Test (Unlabeled Soundscapes) SNR Distribution (Raw vs Preprocessed)\nplt.figure(figsize=(12, 6))\nsns.kdeplot(test_snrs_raw, color='crimson', label='Raw Test SNR', fill=True, alpha=0.3)\nsns.kdeplot(test_snrs_prep, color='dodgerblue', label='Preprocessed Test SNR (ASGP)', fill=True, alpha=0.4)\nplt.title('Unlabeled Soundscapes Test Data SNR Distribution: Raw vs. ASGP Preprocessed')\nplt.xlabel('Estimated SNR (dB)')\nplt.ylabel('Density')\nplt.legend()\nplt.show()\n\n# Plot 5: Overlapping Density Comparison of Train vs Test SNR (Preprocessed)\nplt.figure(figsize=(12, 6))\nsns.kdeplot(part_df['snr_prep'].dropna(), color='forestgreen', label='Train Data (ASGP)', fill=True, alpha=0.3)\nsns.kdeplot(test_snrs_prep, color='darkorange', label='Unlabeled Test Data (ASGP)', fill=True, alpha=0.3)\nplt.title('SNR Density Comparison: Train Data vs. Unlabeled Test Data (ASGP Preprocessed)')\nplt.xlabel('Preprocessed SNR (dB)')\nplt.ylabel('Density')\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-21T12:21:19.929274Z","iopub.execute_input":"2026-06-21T12:21:19.929571Z","iopub.status.idle":"2026-06-21T12:21:20.798249Z","shell.execute_reply.started":"2026-06-21T12:21:19.929548Z","shell.execute_reply":"2026-06-21T12:21:20.797343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot 6: Species-wise average SNR bar chart (Top 10 species)\ntop_species = part_df['primary_label'].value_counts().nlargest(10).index\nspecies_df = part_df[part_df['primary_label'].isin(top_species)].dropna(subset=['snr_prep'])\n\nplt.figure(figsize=(14, 7))\nsns.boxplot(data=species_df, x='primary_label', y='snr_prep', palette='Set3')\nsns.stripplot(data=species_df, x='primary_label', y='snr_prep', color='black', alpha=0.3, jitter=0.2)\nplt.title('Preprocessed SNR Distribution Across Top 10 Most Common Bird Species in Training Sample')\nplt.xlabel('Bird Species (Primary Label)')\nplt.ylabel('Preprocessed SNR (dB)')\nplt.xticks(rotation=45)\nplt.tight_layout()\nplt.show()\n\n# Let's print out the summary statistics of SNR improvement\nimprovement = part_df['snr_prep'] - part_df['snr_raw']\nprint(\"=== PREPROCESSING SYSTEM PERFORMANCE ===\")\nprint(f\"Mean Raw Train SNR: {part_df['snr_raw'].mean():.2f} dB\")\nprint(f\"Mean Preprocessed Train SNR: {part_df['snr_prep'].mean():.2f} dB\")\nprint(f\"Mean SNR Improvement via ASGP: {improvement.mean():.2f} dB\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-21T12:21:20.799472Z","iopub.execute_input":"2026-06-21T12:21:20.799809Z","iopub.status.idle":"2026-06-21T12:21:21.317415Z","shell.execute_reply.started":"2026-06-21T12:21:20.799775Z","shell.execute_reply":"2026-06-21T12:21:21.316436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === PRINT FULL DATASET RESULTS AND STATISTICAL SUMMARY ===\nimport numpy as np\nimport pandas as pd\nfrom scipy.stats import pearsonr, spearmanr\nprint(\"==================================================================\")\nprint(\"             BIOACOUSTIC DENOISING SYSTEM REPORT                  \")\nprint(\"==================================================================\")\n# 1. Training Dataset (1000 samples)\nprint(\"\\n--- 1. TRAINING DATASET STATISTICAL SUMMARY (1000 Samples) ---\")\ntrain_raw_mean = part_df['snr_raw'].mean()\ntrain_raw_std = part_df['snr_raw'].std()\ntrain_prep_mean = part_df['snr_prep'].mean()\ntrain_prep_std = part_df['snr_prep'].std()\ntrain_improvement = part_df['snr_prep'] - part_df['snr_raw']\nprint(f\"Mean Raw Train SNR:          {train_raw_mean:.4f} dB (± {train_raw_std:.4f})\")\nprint(f\"Mean Preprocessed Train SNR: {train_prep_mean:.4f} dB (± {train_prep_std:.4f})\")\nprint(f\"Average SNR Improvement:     {train_improvement.mean():.4f} dB\")\nprint(f\"Max SNR Improvement:         {train_improvement.max():.4f} dB\")\nprint(f\"Min SNR Improvement:         {train_improvement.min():.4f} dB\")\n# 2. Metadata Rating vs. SNR Correlation\nprint(\"\\n--- 2. METADATA RATING VS. SNR CORRELATION ---\")\nvalid_df = part_df.dropna(subset=['snr_prep', 'rating'])\np_corr, p_pval = pearsonr(valid_df['rating'], valid_df['snr_prep'])\ns_corr, s_pval = spearmanr(valid_df['rating'], valid_df['snr_prep'])\nprint(f\"Pearson Correlation (Rating vs. Preprocessed SNR): {p_corr:.4f} (p-value: {p_pval:.2e})\")\nprint(f\"Spearman Correlation (Rating vs. Preprocessed SNR): {s_corr:.4f} (p-value: {s_pval:.2e})\")\n# Grouped by rating\nprint(\"\\nMean SNR by Rating Group:\")\nrating_summary = part_df.groupby('rating')[['snr_raw', 'snr_prep']].agg(['mean', 'std', 'count'])\nprint(rating_summary.to_string())\n# 3. Species-wise Performance (Top 10 Species)\nprint(\"\\n--- 3. TOP 10 SPECIES SNR SUMMARY ---\")\ntop_species = part_df['primary_label'].value_counts().nlargest(10).index\nspecies_df = part_df[part_df['primary_label'].isin(top_species)].copy()\nspecies_df['improvement'] = species_df['snr_prep'] - species_df['snr_raw']\nspecies_summary = species_df.groupby('primary_label')[['snr_raw', 'snr_prep', 'improvement']].mean()\nprint(species_summary.to_string())\n# 4. Test Dataset (Unlabeled Soundscapes)\nprint(\"\\n--- 4. TEST DATASET STATISTICAL SUMMARY (Unlabeled Soundscapes) ---\")\nif 'test_snrs_raw' in locals() and len(test_snrs_raw) > 0:\n    test_raw_arr = np.array(test_snrs_raw)\n    test_prep_arr = np.array(test_snrs_prep)\n    test_imp_arr = test_prep_arr - test_raw_arr\n    print(f\"Total Processed Test Files: {len(test_raw_arr)}\")\n    print(f\"Mean Raw Test SNR:           {test_raw_arr.mean():.4f} dB (± {test_raw_arr.std():.4f})\")\n    print(f\"Mean Preprocessed Test SNR:  {test_prep_arr.mean():.4f} dB (± {test_prep_arr.std():.4f})\")\n    print(f\"Average Test SNR Improvement: {test_imp_arr.mean():.4f} dB\")\nelse:\n    print(\"Test data statistics not found or not processed.\")\nprint(\"==================================================================\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-06-21T12:21:21.318434Z","iopub.execute_input":"2026-06-21T12:21:21.318707Z","iopub.status.idle":"2026-06-21T12:21:21.359614Z","shell.execute_reply.started":"2026-06-21T12:21:21.318685Z","shell.execute_reply":"2026-06-21T12:21:21.358732Z"}},"outputs":[],"execution_count":null}]}