{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11053663,"sourceType":"datasetVersion","datasetId":6886569},{"sourceId":11075844,"sourceType":"datasetVersion","datasetId":6902807}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport logging\nimport random\nimport gc\nimport glob\nimport time\nimport cv2\nimport math\nimport warnings\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import roc_auc_score\nimport librosa\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom tqdm.auto import tqdm\n\nimport timm\n\nwarnings.filterwarnings(\"ignore\")\nlogging.basicConfig(level=logging.ERROR)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-20T03:10:33.045075Z","iopub.execute_input":"2025-05-20T03:10:33.045813Z","iopub.status.idle":"2025-05-20T03:10:55.521335Z","shell.execute_reply.started":"2025-05-20T03:10:33.045772Z","shell.execute_reply":"2025-05-20T03:10:55.520509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CFG:\n    \n    seed = 42\n    debug = False\n    apex = False\n    print_freq = 100\n    num_workers = 2\n    \n    OUTPUT_DIR = '/kaggle/working/'\n\n    train_datadir = '/kaggle/input/birdclef-2025/train_audio'\n    train_csv = '/kaggle/input/birdclef-2025/train.csv'\n    train_soundscapes = '/kaggle/input/birdclef-2025/train_soundscapes'\n    test_soundscapes = '/kaggle/input/birdclef-2025/test_soundscapes'\n    submission_csv = '/kaggle/input/birdclef-2025/sample_submission.csv'\n    taxonomy_csv = '/kaggle/input/birdclef-2025/taxonomy.csv'\n\n    spectrogram_npy = '/kaggle/input/birdclef25-mel-spectrograms/birdclef2025_melspec_5sec_256_256.npy'\n\n\n    human_voice_data = '/kaggle/input/bc25-separation-voice-from-data-by-silero-vad/train_voice_data.pkl'\n    \n    model_name = 'efficientnet_b0'  \n    pretrained = True\n    in_channels = 1\n\n    FS = 32000\n    TARGET_DURATION = 5.0\n    TARGET_SHAPE = (256, 256)\n\n    N_FFT = 1024\n    HOP_LENGTH = 512\n    N_MELS = 128\n    FMIN = 50\n    FMAX = 14000  \n    #N_FFT = 2048\n    #HOP_LENGTH = 128\n    #N_MELS = 512\n    #FMIN = 50\n    #FMAX = 16000\n    \n    device = 'cuda' if torch.cuda.is_available() else 'cpu'\n    epochs = 10  \n    batch_size = 64\n    criterion = 'BCEWithLogitsLoss'\n\n    n_fold = 5\n    selected_folds = [0, 1, 2, 3, 4]   \n\n    optimizer = 'AdamW'\n    lr = 5e-4 \n    weight_decay = 1e-5\n  \n    scheduler = 'CosineAnnealingLR'\n    min_lr = 1e-6\n    T_max = epochs\n\n    aug_prob = 0.5  \n    mixup_alpha = 0.5  \n\n    crop_mode = \"random\"\n    human_threshold = 0.5\n    \n    def update_debug_settings(self):\n        if self.debug:\n            self.epochs = 2\n            self.selected_folds = [0]\n\ncfg = CFG()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T03:10:55.522628Z","iopub.execute_input":"2025-05-20T03:10:55.523161Z","iopub.status.idle":"2025-05-20T03:10:55.534824Z","shell.execute_reply.started":"2025-05-20T03:10:55.523136Z","shell.execute_reply":"2025-05-20T03:10:55.533825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prepare_working_df(cfg):\n    taxonomy_df = pd.read_csv(cfg.taxonomy_csv)\n    species_class_map = dict(zip(taxonomy_df['primary_label'], taxonomy_df['class_name']))\n    train_df = pd.read_csv(cfg.train_csv)\n    \n    label_list = sorted(train_df['primary_label'].unique())\n    label_id_list = list(range(len(label_list)))\n    label2id = dict(zip(label_list, label_id_list))\n    id2label = dict(zip(label_id_list, label_list))\n\n    working_df = train_df[['primary_label', 'rating', 'filename']].copy()\n    working_df['target'] = working_df.primary_label.map(label2id)\n    working_df['filepath'] = cfg.train_datadir + '/' + working_df.filename\n    working_df['samplename'] = working_df.filename.map(lambda x: x.split('/')[0] + '-' + x.split('/')[-1].split('.')[0])\n    working_df['class'] = working_df.primary_label.map(lambda x: species_class_map.get(x, 'Unknown'))\n    \n    return working_df\n\n\ndef audio2melspec(audio_data, cfg):\n    \"\"\"Convert audio data to mel spectrogram\"\"\"\n    if np.isnan(audio_data).any():\n        mean_signal = np.nanmean(audio_data)\n        audio_data = np.nan_to_num(audio_data, nan=mean_signal)\n\n    mel_spec = librosa.feature.melspectrogram(\n        y=audio_data,\n        sr=cfg.FS,\n        n_fft=cfg.N_FFT,\n        hop_length=cfg.HOP_LENGTH,\n        n_mels=cfg.N_MELS,\n        fmin=cfg.FMIN,\n        fmax=cfg.FMAX,\n        power=2.0\n    )\n\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n    mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n    \n    return mel_spec_norm\n\n\ndef remove_human_voice(audio_data, voice_times, sr):\n    \"\"\"Removes human voice segments from the audio data safely.\"\"\"\n    keep_segments = []\n    last_end = 0\n    for segment in voice_times:\n        start = round(segment['start'] * sr)\n        end = round(segment['end'] * sr)\n        # Keep segment before this human voice part\n        keep_segments.append(audio_data[last_end:start])\n        last_end = end\n    # Add the last remaining part after final voice segment\n    keep_segments.append(audio_data[last_end:])\n    return np.concatenate(keep_segments)\n\ndef discard_audio_with_human(audio_path, human_voice_data, start_idx,\n                             end_idx, sr, threshold=0.5):\n    \"\"\"\n    Returns True if human voice portion in the given segment exceeds the threshold ratio.\n    \"\"\"\n    if audio_path not in human_voice_data:\n        return False\n\n    voice_segments = human_voice_data[audio_path]\n    voice_duration = 0.0\n    segment_length = end_idx - start_idx\n\n    for seg in voice_segments:\n        voice_start = seg['start'] * sr\n        voice_end = seg['end'] * sr\n        # Check overlap with segment\n        overlap_start = max(start_idx, voice_start)\n        overlap_end = min(end_idx, voice_end)\n        if overlap_end > overlap_start:\n            voice_duration += (overlap_end - overlap_start)\n\n    voice_ratio = voice_duration / segment_length\n    return voice_ratio > threshold\n\ndef find_all_rms_segments(audio_data, target_samples, step_size):\n    \"\"\"\n    Slide over audio and return list of (start_idx, rms) sorted by rms descending\n    \"\"\"\n    rms_segments = []\n    for start_idx in range(0, len(audio_data) - target_samples + 1, step_size):\n        segment = audio_data[start_idx : start_idx + target_samples]\n        rms = np.sqrt(np.mean(segment ** 2))\n        rms_segments.append((start_idx, rms))\n    # Sort by RMS descending\n    rms_segments.sort(key=lambda x: x[1], reverse=True)\n    return rms_segments\n\ndef process_audio_file(audio_path, human_voice_data, cfg):\n    \"\"\"Process a single audio file to get the mel spectrogram\"\"\"\n    try:\n        audio_data, _ = librosa.load(audio_path, sr=cfg.FS)\n        target_samples = int(cfg.TARGET_DURATION * cfg.FS)\n\n        if len(audio_data) < target_samples:\n            n_copy = math.ceil(target_samples / len(audio_data))\n            if n_copy > 1:\n                audio_data = np.concatenate([audio_data] * n_copy)\n\n        # Select segment based on mode\n        if cfg.crop_mode == \"rms\":\n            step_size = int(cfg.FS)\n            rms_segments = find_all_rms_segments(audio_data, target_samples, step_size)\n\n            start_idx = None\n            for candidate_start, _ in rms_segments:\n                end_idx = candidate_start + target_samples\n                if human_voice_data is not None:\n                    if discard_audio_with_human(audio_path, human_voice_data, candidate_start, end_idx, cfg.FS, cfg.human_threshold):\n                        continue  # discard, try next best segment\n                start_idx = candidate_start\n                break\n            if start_idx is None:\n                # No segment without too much human voice found\n                return None\n        elif cfg.crop_mode == \"random\":\n            max_offset = len(audio_data) - target_samples\n            start_idx = random.randint(0, max(0, max_offset))\n            end_idx = start_idx + target_samples\n            if human_voice_data is not None:\n                if discard_audio_with_human(audio_path, human_voice_data, start_idx, end_idx, cfg.FS,\n                                           cfg.human_threshold):\n                    return None\n        else:  # mode == \"first\"\n            start_idx = 0\n            end_idx = start_idx + target_samples\n            if human_voice_data is not None:\n                if discard_audio_with_human(audio_path, human_voice_data, start_idx, end_idx, cfg.FS,\n                                           cfg.human_threshold):\n                    return None\n\n        cropped_audio = audio_data[start_idx:end_idx]\n\n        if len(cropped_audio) < target_samples:\n            cropped_audio = np.pad(cropped_audio, \n                                 (0, target_samples - len(cropped_audio)), \n                                 mode='constant')\n\n        mel_spec = audio2melspec(cropped_audio, cfg)\n        \n        if mel_spec.shape != cfg.TARGET_SHAPE:\n            mel_spec = cv2.resize(mel_spec, cfg.TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n\n        return mel_spec.astype(np.float32)\n        \n    except Exception as e:\n        print(f\"Error processing {audio_path}: {e}\")\n        return None\n        \n    except Exception as e:\n        print(f\"Error processing {audio_path}: {e}\")\n        return None\n\ndef save_discarded_filenames(filepaths):\n    with open(\"discarded_files.txt\", \"w\") as f:\n        for filepath in filepaths:\n            f.write(filepath + \"\\n\")\n    print(f\"Saved {len(filepaths)} discarded file paths to 'discarded_files.txt'\")\n\ndef generate_spectrograms(cfg):\n    \"\"\"Generate spectrograms from audio files\"\"\"\n    print(\"Generating mel spectrograms from audio files...\")\n    start_time = time.time()\n\n    all_bird_data = {}\n    discarded_files = []\n    errors = []\n\n    df = prepare_working_df(cfg)\n    human_voice_data = pd.read_pickle(cfg.human_voice_data)\n\n    for i, row in tqdm(df.iterrows(), total=len(df)):\n        if cfg.debug and i >= 100:\n            break\n        \n        try:\n            samplename = row['samplename']\n            filepath = row['filepath']\n            \n            mel_spec = process_audio_file(filepath, human_voice_data, cfg)\n            \n            if mel_spec is not None:\n                all_bird_data[samplename] = mel_spec\n            else:\n                discarded_files.append(filepath)\n\n        except Exception as e:\n            print(f\"Error processing {row.filepath}: {e}\")\n            errors.append((row.filepath, str(e)))\n\n    end_time = time.time()\n    print(f\"Processing completed in {end_time - start_time:.2f} seconds\")\n    print(f\"Successfully processed {len(all_bird_data)} files out of {len(df)}\")\n    print(f\"Failed to process {len(errors)} files\")\n\n    save_discarded_filenames(discarded_files)\n\n    return all_bird_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T03:10:55.535884Z","iopub.execute_input":"2025-05-20T03:10:55.536251Z","iopub.status.idle":"2025-05-20T03:10:55.566547Z","shell.execute_reply.started":"2025-05-20T03:10:55.536197Z","shell.execute_reply":"2025-05-20T03:10:55.565538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cfg = CFG()\ncfg.crop_mode = \"rms\"\nbird_data = generate_spectrograms(cfg)\nnp.save(\"birdclef2025_melspec_5sec_256_256_nohuman_rms.npy\", bird_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T03:10:55.568690Z","iopub.execute_input":"2025-05-20T03:10:55.568975Z","iopub.status.idle":"2025-05-20T03:11:17.011825Z","shell.execute_reply.started":"2025-05-20T03:10:55.568950Z","shell.execute_reply":"2025-05-20T03:11:17.010543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def save_pseudo_filenames(filepaths):\n    with open(\"pseudo_files.txt\", \"w\") as f:\n        for filepath in filepaths:\n            f.write(filepath + \"\\n\")\n    print(f\"Saved {len(filepaths)} file paths to 'pseudo_files.txt'\")\n\ndef generate_spectrograms_pseudo(cfg):\n    \"\"\"Generate spectrograms from pseudo audio files\"\"\"\n    print(\"Generating mel spectrograms from pseudo audio files...\")\n    start_time = time.time()\n\n    all_bird_data = {}\n    pseudo_files = []\n    errors = []\n\n    audio_files = glob.glob(os.path.join(cfg.train_soundscapes, '**', '*.ogg'), recursive=True)\n\n    for i, filepath in tqdm(enumerate(audio_files), total=len(audio_files)):\n        if cfg.debug and i >= 100:\n            break\n        \n        try:\n            mel_spec = process_audio_file(filepath, None, cfg)\n            \n            if mel_spec is not None:\n                all_bird_data[filepath] = mel_spec\n                pseudo_files.append(filepath)\n\n        except Exception as e:\n            print(f\"Error processing {filepath}: {e}\")\n            errors.append((filepath, str(e)))\n\n    end_time = time.time()\n    print(f\"Processing completed in {end_time - start_time:.2f} seconds\")\n    print(f\"Successfully processed {len(all_bird_data)} files out of {len(audio_files)}\")\n    print(f\"Failed to process {len(errors)} files\")\n\n    save_pseudo_filenames(pseudo_files)\n    \n    return all_bird_data\n\n#cfg = CFG()\n#cfg.crop_mode = \"random\"\n#pseudo_data = generate_spectrograms_pseudo(cfg)\n#np.save(\"birdclef2025_melspec_5sec_256_256_pseudo.npy\", pseudo_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-20T03:11:17.012488Z","iopub.status.idle":"2025-05-20T03:11:17.012764Z","shell.execute_reply.started":"2025-05-20T03:11:17.012641Z","shell.execute_reply":"2025-05-20T03:11:17.012652Z"}},"outputs":[],"execution_count":null}]}