{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11075844,"sourceType":"datasetVersion","datasetId":6902807}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport librosa\nimport librosa.display\nimport torch\nimport matplotlib.pyplot as plt\nimport soundfile as sf # Silero VAD uses soundfile\n\nclass CFG:\n    # Paths (ADJUST THESE if necessary)\n    train_csv = r'/kaggle/input/birdclef-2025/train.csv' # Example path, adjust!\n    train_datadir = r'/kaggle/input/birdclef-2025/train_audio' # Example path, adjust!\n\n    # Audio parameters (Copy from your baseline script)\n    FS = 32000\n    TARGET_DURATION = 5.0 # Duration used for feature extraction\n    TARGET_SHAPE = (256, 256) # Target shape for resizing\n\n    N_FFT = 1024\n    HOP_LENGTH = 500 # Adjust if different in your processing\n    N_MELS = 128\n    FMIN = 40\n    FMAX = 15000\n\n    # VAD parameters\n    VAD_THRESHOLD = 0.4 # Threshold mentioned in discussion (default is 0.5)\n\ncfg = CFG()\n\n# --- Helper Functions (Adapted from baseline) ---\n\ndef audio_to_melspec(audio_data, cfg):\n    \"\"\"Convert audio data to mel spectrogram\"\"\"\n    if np.isnan(audio_data).any():\n        mean_signal = np.nanmean(audio_data)\n        audio_data = np.nan_to_num(audio_data, nan=mean_signal)\n\n    mel_spec = librosa.feature.melspectrogram(\n        y=audio_data,\n        sr=cfg.FS,\n        n_fft=cfg.N_FFT,\n        hop_length=cfg.HOP_LENGTH,\n        n_mels=cfg.N_MELS,\n        fmin=cfg.FMIN,\n        fmax=cfg.FMAX,\n        power=2.0\n    )\n\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n    # Normalize for visualization (optional, but often helpful)\n    mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n\n    return mel_spec_db # Return dB for better visualization range\n\ndef get_vad_timestamps(audio_path, vad_model_utils, cfg):\n    \"\"\"Get speech timestamps using Silero VAD\"\"\"\n    try:\n        # Silero VAD expects 16kHz, read_audio handles resampling\n        wav = vad_model_utils['read_audio'](audio_path, sampling_rate=16000)\n        speech_timestamps = vad_model_utils['get_speech_timestamps'](\n            wav,\n            vad_model_utils['model'],\n            threshold=cfg.VAD_THRESHOLD,\n            sampling_rate=16000 # Ensure VAD uses 16kHz\n        )\n        return speech_timestamps, 16000 # Return timestamps and the sample rate used by VAD\n    except Exception as e:\n        print(f\"  Error getting VAD timestamps for {os.path.basename(audio_path)}: {e}\")\n        return [], 16000\n\ndef plot_spectrogram_with_vad(spec, vad_timestamps, vad_sr, title, cfg):\n    \"\"\"Plots spectrogram and overlays VAD segments\"\"\"\n    plt.figure(figsize=(12, 5))\n    librosa.display.specshow(spec, sr=cfg.FS, hop_length=cfg.HOP_LENGTH,\n                             x_axis='time', y_axis='mel', fmin=cfg.FMIN, fmax=cfg.FMAX)\n    plt.colorbar(format='%+2.0f dB')\n    plt.title(title)\n\n    # Overlay VAD timestamps\n    # Convert VAD sample indices (at vad_sr) to time (seconds)\n    vad_times = [(ts['start'] / vad_sr, ts['end'] / vad_sr) for ts in vad_timestamps]\n\n    for start_time, end_time in vad_times:\n        plt.axvspan(start_time, end_time, color='red', alpha=0.3, label='Human Voice (VAD)' if 'Human Voice (VAD)' not in plt.gca().get_legend_handles_labels()[1] else \"\")\n\n    if vad_times:\n        plt.legend(loc='upper right')\n    plt.tight_layout()\n    plt.show()\n\n# --- Main Analysis Logic ---\n\n# 1. List of problematic samples from your log output\nproblematic_samples = [\n    \"greegr-XC490733\", \"cinbec1-XC389682\", \"whtdov-iNat863170\",\n    \"bkmtou1-iNat1270065\", \"speowl1-XC719818\", \"mastit1-iNat227747\",\n    \"amekes-iNat863228\", \"amekes-iNat522505\", \"saffin-iNat157525\",\n    \"grekis-XC360620\"\n    # Add more if needed, removed duplicates from your log example\n]\nprint(f\"Analyzing {len(problematic_samples)} problematic samples...\")\n\n# 2. Load metadata\ntry:\n    train_df = pd.read_csv(cfg.train_csv)\n    # Create samplename if it doesn't exist (match training script logic)\n    if 'samplename' not in train_df.columns:\n         train_df['samplename'] = train_df.filename.map(lambda x: x.split('/')[0] + '-' + x.split('/')[-1].split('.')[0])\n    # Create filepath if it doesn't exist\n    if 'filepath' not in train_df.columns:\n        train_df['filepath'] = train_df['filename'].apply(lambda x: os.path.join(cfg.train_datadir, x))\n\n    # Create mapping for quick lookup\n    samplename_to_filepath = pd.Series(train_df.filepath.values, index=train_df.samplename).to_dict()\n    print(f\"Loaded metadata for {len(train_df)} samples.\")\nexcept FileNotFoundError:\n    print(f\"Error: Could not find train.csv at {cfg.train_csv}\")\n    exit()\nexcept Exception as e:\n    print(f\"Error loading metadata: {e}\")\n    exit()\n\n\n# 3. Load VAD model\ntry:\n    # Use force_reload=True if you update the library or encounter issues\n    model, utils = torch.hub.load(repo_or_dir='snakers4/silero-vad',\n                                  model='silero_vad',\n                                  force_reload=False) # Set to True if needed\n    vad_model_utils = {\n        'model': model,\n        'get_speech_timestamps': utils[0],\n        'save_audio': utils[1],\n        'read_audio': utils[2],\n        'VADIterator': utils[3],\n        'collect_chunks': utils[4]\n    }\n    print(\"Silero VAD model loaded successfully.\")\nexcept Exception as e:\n    print(f\"Error loading Silero VAD model: {e}\")\n    print(\"Please ensure you have internet connectivity and the model can be downloaded.\")\n    exit()\n\n# 4. Analyze each sample\nfor sample in problematic_samples:\n    print(f\"\\n--- Processing: {sample} ---\")\n    if sample not in samplename_to_filepath:\n        print(f\"  Warning: Samplename '{sample}' not found in train.csv metadata.\")\n        continue\n\n    audio_path = samplename_to_filepath[sample]\n\n    if not os.path.exists(audio_path):\n        print(f\"  Error: Audio file not found at '{audio_path}'\")\n        continue\n\n    try:\n        # Load full audio first\n        audio_data, _ = librosa.load(audio_path, sr=cfg.FS, mono=True)\n        print(f\"  Loaded audio: Duration={librosa.get_duration(y=audio_data, sr=cfg.FS):.2f}s\")\n\n        # Generate spectrogram for the *entire* audio for context\n        # (Your processing likely took a 5s chunk, but seeing the whole helps)\n        full_spec = audio_to_melspec(audio_data, cfg)\n\n        # Get VAD timestamps for the full audio\n        vad_timestamps, vad_sr = get_vad_timestamps(audio_path, vad_model_utils, cfg)\n        print(f\"  Detected {len(vad_timestamps)} potential speech segments.\")\n\n        # Plot\n        plot_spectrogram_with_vad(full_spec, vad_timestamps, vad_sr,\n                                  f\"Full Spectrogram & VAD: {sample}\", cfg)\n\n    except Exception as e:\n        print(f\"  Error processing sample {sample}: {e}\")\n\nprint(\"\\nAnalysis complete.\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-17T17:19:11.202642Z","iopub.execute_input":"2025-04-17T17:19:11.202874Z","iopub.status.idle":"2025-04-17T17:19:46.815718Z","shell.execute_reply.started":"2025-04-17T17:19:11.202851Z","shell.execute_reply":"2025-04-17T17:19:46.814615Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport librosa\nimport librosa.display\nimport numpy as np\nimport soundfile as sf\nimport matplotlib.pyplot as plt\n\n# --- Configuration ---\nclass CFG:\n    # Paths (ADJUST THESE)\n    train_csv = r'/kaggle/input/birdclef-2025/train.csv' # Path to your train CSV\n    train_datadir = r'/kaggle/input/birdclef-2025/train_audio' # Path to your train audio folder\n    output_dir = r'/kaggle/working/' # Where to save filtered audio\n\n    # Audio parameters\n    FS = 32000 # Sample rate\n\n    # Filtering parameters\n    filter_type = 'highpass' # 'highpass', 'lowpass', 'bandstop'\n    cutoff_freq_highpass = 1500 # Hz - Frequencies below this will be attenuated\n    # cutoff_freq_lowpass = 800 # Hz - Frequencies above this will be attenuated\n    # bandstop_low = 80 # Hz\n    # bandstop_high = 1100 # Hz\n\n    # Sample to process (Choose one from your problematic list)\n    sample_filename = 'speowl1/XC719818.ogg' # Example using greegr-XC490733 bkmtou1-iNat1270065 speowl1-XC719818\n\ncfg = CFG()\n\n# --- Helper Functions ---\ndef apply_fft_filter(y, sr, filter_type, cutoff_low=None, cutoff_high=None):\n    \"\"\"Applies a filter in the frequency domain using FFT.\"\"\"\n    # Compute the Short-Time Fourier Transform (STFT)\n    stft_result = librosa.stft(y)\n    # Get the frequency bins corresponding to the STFT\n    freqs = librosa.fft_frequencies(sr=sr) # Get frequencies for each bin\n\n    # Create a copy to modify\n    stft_filtered = stft_result.copy()\n\n    print(f\"Applying {filter_type} filter...\")\n    if filter_type == 'highpass':\n        if cutoff_low is None:\n            raise ValueError(\"cutoff_low must be specified for highpass filter\")\n        print(f\"  Attenuating frequencies below {cutoff_low} Hz\")\n        # Zero out bins corresponding to frequencies below the cutoff\n        stft_filtered[freqs < cutoff_low, :] = 0\n    elif filter_type == 'lowpass':\n        if cutoff_high is None:\n            raise ValueError(\"cutoff_high must be specified for lowpass filter\")\n        print(f\"  Attenuating frequencies above {cutoff_high} Hz\")\n        # Zero out bins corresponding to frequencies above the cutoff\n        stft_filtered[freqs > cutoff_high, :] = 0\n    elif filter_type == 'bandstop':\n        if cutoff_low is None or cutoff_high is None:\n            raise ValueError(\"cutoff_low and cutoff_high must be specified for bandstop filter\")\n        print(f\"  Attenuating frequencies between {cutoff_low} Hz and {cutoff_high} Hz\")\n        # Find indices for frequencies within the band\n        band_indices = np.where((freqs >= cutoff_low) & (freqs <= cutoff_high))[0]\n        # Zero out bins within the band\n        stft_filtered[band_indices, :] = 0\n    else:\n        raise ValueError(f\"Unknown filter_type: {filter_type}\")\n\n    # Compute the Inverse Short-Time Fourier Transform (ISTFT)\n    y_filtered = librosa.istft(stft_filtered, length=len(y)) # Ensure output length matches input\n\n    return y_filtered\n\ndef plot_spectrogram(y, sr, title, ax):\n    \"\"\"Helper to plot a spectrogram.\"\"\"\n    S = librosa.feature.melspectrogram(y=y, sr=sr)\n    S_db = librosa.power_to_db(S, ref=np.max)\n    img = librosa.display.specshow(S_db, sr=sr, x_axis='time', y_axis='mel', ax=ax)\n    ax.set_title(title)\n    return img\n\n# --- Main Script ---\nif __name__ == \"__main__\":\n    # Ensure output directory exists\n    os.makedirs(cfg.output_dir, exist_ok=True)\n\n    # Construct full path to the audio file\n    audio_path = os.path.join(cfg.train_datadir, cfg.sample_filename)\n\n    if not os.path.exists(audio_path):\n        print(f\"Error: Audio file not found at {audio_path}\")\n        exit()\n\n    print(f\"Loading audio: {audio_path}\")\n    try:\n        y, sr = librosa.load(audio_path, sr=cfg.FS)\n        print(f\"Audio loaded successfully. Duration: {librosa.get_duration(y=y, sr=sr):.2f}s\")\n\n        # --- Apply the filter ---\n        y_filtered = apply_fft_filter(\n            y, sr,\n            filter_type=cfg.filter_type,\n            cutoff_low=cfg.cutoff_freq_highpass if cfg.filter_type in ['highpass', 'bandstop'] else None,\n            cutoff_high=None # Adjust if using lowpass or bandstop\n            # cutoff_high=cfg.cutoff_freq_lowpass if cfg.filter_type == 'lowpass' else (cfg.bandstop_high if cfg.filter_type == 'bandstop' else None) # More general\n        )\n        print(\"Filtering complete.\")\n\n        # --- Save audio files ---\n        original_filename = f\"original_{os.path.splitext(os.path.basename(audio_path))[0]}.wav\"\n        filtered_filename = f\"filtered_{cfg.filter_type}{cfg.cutoff_freq_highpass}_{os.path.splitext(os.path.basename(audio_path))[0]}.wav\" # Example naming\n\n        original_save_path = os.path.join(cfg.output_dir, original_filename)\n        filtered_save_path = os.path.join(cfg.output_dir, filtered_filename)\n\n        print(f\"Saving original audio to: {original_save_path}\")\n        sf.write(original_save_path, y, sr)\n\n        print(f\"Saving filtered audio to: {filtered_save_path}\")\n        sf.write(filtered_save_path, y_filtered, sr)\n\n        # --- Plot Spectrograms for Comparison ---\n        fig, axs = plt.subplots(2, 1, figsize=(10, 8), sharex=True, sharey=True)\n\n        plot_spectrogram(y, sr, \"Original Spectrogram\", axs[0])\n        img = plot_spectrogram(y_filtered, sr, f\"Filtered Spectrogram ({cfg.filter_type} @ {cfg.cutoff_freq_highpass}Hz)\", axs[1])\n\n        fig.colorbar(img, ax=axs, format=\"%+2.0f dB\")\n        plt.tight_layout()\n        plot_save_path = os.path.join(cfg.output_dir, f\"spectrogram_comparison_{os.path.splitext(os.path.basename(audio_path))[0]}.png\")\n        plt.savefig(plot_save_path)\n        print(f\"Saved spectrogram comparison plot to: {plot_save_path}\")\n        plt.show()\n\n        print(\"\\nFinished. Please listen to the saved .wav files in:\")\n        print(f\"{cfg.output_dir}\")\n\n    except Exception as e:\n        print(f\"An error occurred: {e}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T17:43:40.862846Z","iopub.execute_input":"2025-04-17T17:43:40.863526Z","iopub.status.idle":"2025-04-17T17:43:43.201913Z","shell.execute_reply.started":"2025-04-17T17:43:40.863496Z","shell.execute_reply":"2025-04-17T17:43:43.200706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}