{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":25954,"databundleVersionId":2091745,"isSourceIdPinned":false},{"sourceType":"competition","sourceId":33246,"databundleVersionId":3221581,"isSourceIdPinned":false},{"sourceType":"competition","sourceId":44224,"databundleVersionId":5188730,"isSourceIdPinned":false},{"sourceType":"competition","sourceId":70203,"databundleVersionId":8068726,"isSourceIdPinned":false},{"sourceType":"competition","sourceId":91844,"databundleVersionId":11361821,"isSourceIdPinned":false},{"sourceType":"competition","sourceId":129329,"databundleVersionId":15996945,"isSourceIdPinned":false}],"dockerImageVersionId":31328,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BirdCLEF 2026 Data Preprocessing Notebook","metadata":{}},{"cell_type":"markdown","source":"This notebook demonstrates how we can transform audio data into mel-spectrogram data. This transformation is essential for training 2D Convolutional Neural Networks (CNNs) on audio data, as it converts the one-dimensional audio signals into two-dimensional image-like representations.","metadata":{}},{"cell_type":"markdown","source":"This notebook will be demonstrating the below preprocessing steps: \n* Resample 32kHz mono\n* Filter rating < 3\n* random crop within the first 6 seconds\n* Save log-mel as .npy","metadata":{}},{"cell_type":"markdown","source":"# Import Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport librosa\nimport ast \nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom sklearn.model_selection import StratifiedKFold\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:33:16.034709Z","iopub.execute_input":"2026-05-01T12:33:16.035198Z","iopub.status.idle":"2026-05-01T12:33:17.484538Z","shell.execute_reply.started":"2026-05-01T12:33:16.035164Z","shell.execute_reply":"2026-05-01T12:33:17.483701Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Config ","metadata":{}},{"cell_type":"code","source":"class CFG:\n\n    ROOT_2026  = \"/kaggle/input/competitions/birdclef-2026\"\n    AUDIO_2026  = f\"{ROOT_2026}/train_audio\"\n    META_2026    = f\"{ROOT_2026}/train.csv\"\n    TAXONOMY    = f\"{ROOT_2026}/taxonomy.csv\"\n    SAVE_DIR    = \"/kaggle/working/train_spectrograms\"\n    META_OUT     = '/kaggle/working/train_metadata_with_folds.csv'\n    INCLUDE_EXTRA = True\n    EXTRA_SOURCES = [\n        {\n            'audio_dir': '/kaggle/input/competitions/birdclef-2025/train_audio',\n            'meta_path': '/kaggle/input/competitions/birdclef-2025/train.csv',\n            'year'     : 2025,\n        },\n        {\n            'audio_dir': '/kaggle/input/competitions/birdclef-2024/train_audio',\n            'meta_path': '/kaggle/input/competitions/birdclef-2024/train_metadata.csv',\n            'year'     : 2024,\n        },\n        {\n            'audio_dir': '/kaggle/input/competitions/birdclef-2023/train_audio',\n            'meta_path': '/kaggle/input/competitions/birdclef-2023/train_metadata.csv',\n            'year'     : 2023,\n        },\n\n        {\n            'audio_dir': '/kaggle/input/competitions/birdclef-2022/train_audio',\n            'meta_path': '/kaggle/input/competitions/birdclef-2022/train_metadata.csv',\n            'year'     : 2022,\n        },\n        {\n            'audio_dir': '/kaggle/input/competitions/birdclef-2021/train_short_audio',\n            'meta_path': '/kaggle/input/competitions/birdclef-2021/train_metadata.csv',\n            'year'     : 2021,\n        },\n\n       \n    ]\n\n    MAX_PER_SPECIES_EXTRA = 500 \n    \n    MIN_RATING   = 3.0 \n\n\n    SR          = 32000      \n    DURATION    = 5         \n    SAMPLES     = SR * DURATION \n    SIX_SEC      = SR * 6 \n\n\n    N_FFT       = 1024\n    HOP_LENGTH  = 320       \n    N_MELS      = 128\n    FMIN        = 20\n    FMAX        = 16000\n\n\n    N_FOLDS     = 5\n    SEED        = 42\n\nos.makedirs(CFG.SAVE_DIR, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:33:37.375872Z","iopub.execute_input":"2026-05-01T12:33:37.376346Z","iopub.status.idle":"2026-05-01T12:33:37.386900Z","shell.execute_reply.started":"2026-05-01T12:33:37.376317Z","shell.execute_reply":"2026-05-01T12:33:37.385614Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(CFG.META_2026)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:34:54.546429Z","iopub.execute_input":"2026-05-01T12:34:54.547756Z","iopub.status.idle":"2026-05-01T12:34:54.747069Z","shell.execute_reply.started":"2026-05-01T12:34:54.547661Z","shell.execute_reply":"2026-05-01T12:34:54.745549Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:34:54.966227Z","iopub.execute_input":"2026-05-01T12:34:54.966743Z","iopub.status.idle":"2026-05-01T12:34:54.985960Z","shell.execute_reply.started":"2026-05-01T12:34:54.966652Z","shell.execute_reply":"2026-05-01T12:34:54.984740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['primary_label'].nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:34:55.215923Z","iopub.execute_input":"2026-05-01T12:34:55.217121Z","iopub.status.idle":"2026-05-01T12:34:55.227009Z","shell.execute_reply.started":"2026-05-01T12:34:55.217077Z","shell.execute_reply":"2026-05-01T12:34:55.225745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:34:55.535301Z","iopub.execute_input":"2026-05-01T12:34:55.535651Z","iopub.status.idle":"2026-05-01T12:34:55.542920Z","shell.execute_reply.started":"2026-05-01T12:34:55.535624Z","shell.execute_reply":"2026-05-01T12:34:55.541610Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.columns.tolist()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:34:55.855605Z","iopub.execute_input":"2026-05-01T12:34:55.856138Z","iopub.status.idle":"2026-05-01T12:34:55.863236Z","shell.execute_reply.started":"2026-05-01T12:34:55.856106Z","shell.execute_reply":"2026-05-01T12:34:55.862129Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"taxonomy = pd.read_csv(CFG.TAXONOMY)\n\nprint('\\nTaxonomy class breakdown:')\nprint(taxonomy['class_name'].value_counts().to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:34:56.095385Z","iopub.execute_input":"2026-05-01T12:34:56.095893Z","iopub.status.idle":"2026-05-01T12:34:56.106872Z","shell.execute_reply.started":"2026-05-01T12:34:56.095861Z","shell.execute_reply":"2026-05-01T12:34:56.105545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"species_counts = df['primary_label'].value_counts()\nlow_quality    = (df['rating'] < CFG.MIN_RATING).sum()\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:34:56.381142Z","iopub.execute_input":"2026-05-01T12:34:56.381622Z","iopub.status.idle":"2026-05-01T12:34:56.391320Z","shell.execute_reply.started":"2026-05-01T12:34:56.381589Z","shell.execute_reply":"2026-05-01T12:34:56.390101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('=== Dataset overview ===')\nprint(f'Total records                    : {len(df):,}')\nprint(f'Species                          : {species_counts.shape[0]}')\nprint(f'Species with < 5 recordings      : {(species_counts < 5).sum()}')\nprint(f'Species with < 30 recordings     : {(species_counts < 30).sum()}')\nprint(f'Median recordings per species    : {species_counts.median():.0f}')\nprint(f'Max recordings per species       : {species_counts.max()}')\nprint(f'Dropped by rating < {CFG.MIN_RATING}          : {low_quality:,} ({low_quality/len(df)*100:.1f}%)')\nprint()\nprint('Rating distribution:')\nprint(df['rating'].value_counts().sort_index().to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:34:56.646000Z","iopub.execute_input":"2026-05-01T12:34:56.646304Z","iopub.status.idle":"2026-05-01T12:34:56.656545Z","shell.execute_reply.started":"2026-05-01T12:34:56.646279Z","shell.execute_reply":"2026-05-01T12:34:56.655300Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Audio to Melspectrogram ","metadata":{}},{"cell_type":"code","source":"def audio_to_melspec(y, sr=CFG.SR):\n\n    mel = librosa.feature.melspectrogram(\n        y=y,\n        sr=sr,\n        n_fft=CFG.N_FFT,\n        hop_length=CFG.HOP_LENGTH,\n        n_mels=CFG.N_MELS,\n        fmin=CFG.FMIN,\n        fmax=CFG.FMAX\n    )\n\n    mel = librosa.power_to_db(mel, ref=np.max)\n\n\n    mel = (mel - mel.min()) / (mel.max() - mel.min() + 1e-6)\n\n    return mel.astype(np.float32)  \n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:34:57.295931Z","iopub.execute_input":"2026-05-01T12:34:57.297165Z","iopub.status.idle":"2026-05-01T12:34:57.303479Z","shell.execute_reply.started":"2026-05-01T12:34:57.297126Z","shell.execute_reply":"2026-05-01T12:34:57.302104Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Random Crop and Filter the rating ","metadata":{}},{"cell_type":"code","source":"def load_and_chunk(filepath, rating=None):\n\n    if rating is not None:\n        try:\n            r = float(rating)\n            if not np.isnan(r) and r < CFG.MIN_RATING:\n                return []\n        except (ValueError, TypeError):\n            pass  \n\n\n    try:\n        y, sr = librosa.load(filepath, sr=CFG.SR, mono=True)\n    except Exception as e:\n        print(f'  [LOAD ERROR] {Path(filepath).name}: {e}')\n        return []\n\n\n    if y.std() < 0.001:\n        return []\n\n    total_samples = len(y)\n    target        = CFG.SAMPLES  \n    chunks        = []\n\n\n    if total_samples < target:\n        y_padded = np.pad(y, (0, target - total_samples), mode='constant')\n        chunks.append(y_padded)\n\n    else:\n\n        first_window = min(total_samples, CFG.SIX_SEC) \n        max_start    = first_window - target             \n        start        = np.random.randint(0, max_start) if max_start > 0 else 0\n        chunks.append(y[start : start + target])\n\n\n        if total_samples > target * 2:\n            chunks.append(y[-target:])\n\n    return chunks\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:34:59.540625Z","iopub.execute_input":"2026-05-01T12:34:59.541868Z","iopub.status.idle":"2026-05-01T12:34:59.551041Z","shell.execute_reply.started":"2026-05-01T12:34:59.541826Z","shell.execute_reply":"2026-05-01T12:34:59.549820Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Secondary labels parser","metadata":{}},{"cell_type":"code","source":"def parse_secondary_labels(raw):\n\n    if raw is None:\n        return []\n    if isinstance(raw, list):\n        return raw\n    if isinstance(raw, float) and np.isnan(raw):\n        return []\n    raw = str(raw).strip()\n    if raw in ('', '[]', 'nan'):\n        return []\n    try:\n        result = ast.literal_eval(raw)\n        return result if isinstance(result, list) else []\n    except Exception:\n        return []\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:35:00.336565Z","iopub.execute_input":"2026-05-01T12:35:00.337867Z","iopub.status.idle":"2026-05-01T12:35:00.345021Z","shell.execute_reply.started":"2026-05-01T12:35:00.337823Z","shell.execute_reply":"2026-05-01T12:35:00.343160Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Metadata standardiser","metadata":{}},{"cell_type":"code","source":"COMMON_COLS = [\n    'primary_label', 'secondary_labels', 'rating',\n    'filename', 'audio_dir', 'year', 'npy_path', 'chunk_id', 'fold',\n]\n\n\nsci_to_2026id = dict(zip(\n    taxonomy['scientific_name'].str.strip().str.lower(),\n    taxonomy['inat_taxon_id'].astype(str)\n))\nvalid_2026_ids = set(taxonomy['inat_taxon_id'].astype(str).unique())\nprint(f\"2026 master lookup    : {len(sci_to_2026id)} species\")\n\n\nEBIRD_PATH = '/kaggle/input/competitions/birdclef-2024/eBird_Taxonomy_v2021.csv'\n\nif os.path.exists(EBIRD_PATH):\n    ebird = pd.read_csv(EBIRD_PATH)\n    ebird_code_to_sci = dict(zip(\n        ebird['SPECIES_CODE'].str.strip().str.lower(),\n        ebird['SCI_NAME'].str.strip().str.lower()\n    ))\n\n    ebird_code_to_2026id = {\n        code: sci_to_2026id[sci]\n        for code, sci in ebird_code_to_sci.items()\n        if sci in sci_to_2026id\n    }\n    print(f\"eBird → 2026 bridge   : {len(ebird_code_to_2026id)} overlapping species\")\n    print(f\"Sample                : {list(ebird_code_to_2026id.items())[:5]}\")\nelse:\n    ebird_code_to_sci    = {}\n    ebird_code_to_2026id = {}\n    print(f\"[WARN] eBird file not found at {EBIRD_PATH}\")\n\n\ndef standardise_df(df, audio_dir, year):\n    df = df.copy()\n\n\n    if year == 2021:\n        df['filename'] = (df['primary_label'].astype(str)\n                          + '/' + df['filename'].astype(str))\n\n\n    if year == 2025:\n        before = len(df)\n        df['primary_label'] = (\n            df['scientific_name'].str.strip().str.lower()\n            .map(sci_to_2026id)\n        )\n        df = df[df['primary_label'].notna()].copy()\n        print(f\"  2025: {before:,} → {len(df):,} rows \"\n              f\"(scientific_name → 2026 inat_taxon_id)\")\n\n\n    if year in (2022, 2023, 2024):\n        before = len(df)\n        df['primary_label'] = (\n            df['primary_label'].astype(str).str.strip().str.lower()\n            .map(ebird_code_to_2026id)\n        )\n        df = df[df['primary_label'].notna()].copy()\n        print(f\"  {year}: {before:,} → {len(df):,} rows \"\n              f\"(eBird code → scientific_name → 2026 inat_taxon_id)\")\n\n\n    if year == 2021:\n        before = len(df)\n        df['primary_label'] = (\n            df['scientific_name'].str.strip().str.lower()\n            .map(sci_to_2026id)\n        )\n        df = df[df['primary_label'].notna()].copy()\n        print(f\"  2021: {before:,} → {len(df):,} rows \"\n              f\"(scientific_name → 2026 inat_taxon_id)\")\n\n\n    if 'secondary_labels' not in df.columns:\n        df['secondary_labels'] = '[]'\n    if 'rating' not in df.columns:\n        df['rating'] = 0.0\n\n    if 'primary_label' not in df.columns:\n        print(f\"  [ERROR] {year}: primary_label missing after transforms.\")\n        return pd.DataFrame()\n\n    df['audio_dir'] = audio_dir\n    df['year']      = year\n    df['npy_path']  = ''\n    df['chunk_id']  = 0\n    df['fold']      = -1\n\n    for col in COMMON_COLS:\n        if col not in df.columns:\n            df[col] = None\n\n    return df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:35:29.296067Z","iopub.execute_input":"2026-05-01T12:35:29.296913Z","iopub.status.idle":"2026-05-01T12:35:29.410875Z","shell.execute_reply.started":"2026-05-01T12:35:29.296874Z","shell.execute_reply":"2026-05-01T12:35:29.409749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_2026 = standardise_df(df, CFG.AUDIO_2026, 2026)\n\nprint(df_2026['primary_label'].dropna().head(10).tolist())\nprint(f\"Unique species: {df_2026['primary_label'].nunique()}\")\n\nvalid_species = set(df_2026['primary_label'].astype(str).unique())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:35:34.335340Z","iopub.execute_input":"2026-05-01T12:35:34.336597Z","iopub.status.idle":"2026-05-01T12:35:34.359564Z","shell.execute_reply.started":"2026-05-01T12:35:34.336550Z","shell.execute_reply":"2026-05-01T12:35:34.358104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"extra_dfs = []\n\nif CFG.INCLUDE_EXTRA:\n    print(f\"Valid 2026 species: {len(valid_species)}\")\n\n    for src in CFG.EXTRA_SOURCES:\n        audio_dir = src['audio_dir']\n        meta_path = src['meta_path']\n        year      = src['year']\n\n        if not os.path.exists(meta_path) or not os.path.exists(audio_dir):\n            print(f\"  [SKIP] {year}: path not found\")\n            continue\n\n        df_yr = pd.read_csv(meta_path)\n        df_yr = standardise_df(df_yr, audio_dir, year)\n\n\n        if df_yr.empty:\n            print(f\"  [SKIP] {year}: no species overlap with 2026\")\n            continue\n\n        n_matched = len(df_yr)\n\n\n        mask_low = df_yr['rating'].notna() & (df_yr['rating'] < CFG.MIN_RATING)\n        df_yr    = df_yr[~mask_low].copy()\n        n_after_rating = len(df_yr)\n\n    \n        df_yr = (\n            df_yr\n            .sort_values('rating', ascending=False, na_position='last')\n            .groupby('primary_label')\n            .head(CFG.MAX_PER_SPECIES_EXTRA)\n            .reset_index(drop=True)\n        )\n\n        print(f\"  {year}: {n_matched:,} matched\"\n              f\" → {n_after_rating:,} after rating filter\"\n              f\" → {len(df_yr):,} after cap\")\n\n        if len(df_yr) > 0:\n            extra_dfs.append(df_yr)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:36:10.815302Z","iopub.execute_input":"2026-05-01T12:36:10.815797Z","iopub.status.idle":"2026-05-01T12:36:12.029822Z","shell.execute_reply.started":"2026-05-01T12:36:10.815763Z","shell.execute_reply":"2026-05-01T12:36:12.028160Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if extra_dfs:\n    df_all = pd.concat([df_2026] + extra_dfs, ignore_index=True)\nelse:\n    df_all = df_2026.copy()\n\ndf_all = df_all.reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:36:32.496221Z","iopub.execute_input":"2026-05-01T12:36:32.496670Z","iopub.status.idle":"2026-05-01T12:36:32.585310Z","shell.execute_reply.started":"2026-05-01T12:36:32.496625Z","shell.execute_reply":"2026-05-01T12:36:32.583820Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=CFG.N_FOLDS, shuffle=True, random_state=CFG.SEED)\n\nfor fold, (_, val_idx) in enumerate(skf.split(df_all, df_all['primary_label'])):\n    df_all.loc[val_idx, 'fold'] = fold\n\n\nspecies_counts = df_all['primary_label'].value_counts()\nrare_species   = set(species_counts[species_counts < CFG.N_FOLDS].index)\ndf_all.loc[df_all['primary_label'].isin(rare_species), 'fold'] = -1\n\nprint(f'Rare species (fold=-1)  : {len(rare_species)}')\nprint(f'\\nFold distribution (before processing):')\nprint(df_all['fold'].value_counts().sort_index().to_string())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:36:37.935814Z","iopub.execute_input":"2026-05-01T12:36:37.936182Z","iopub.status.idle":"2026-05-01T12:36:38.030143Z","shell.execute_reply.started":"2026-05-01T12:36:37.936152Z","shell.execute_reply":"2026-05-01T12:36:38.029138Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Main Processing ","metadata":{}},{"cell_type":"code","source":"failed_files  = []   \nextra_rows    = []   \n\nfor idx, row in tqdm(df_all.iterrows(), total=len(df_all), desc='Processing audio'):\n\n    filepath = os.path.join(str(row['audio_dir']), str(row['filename']))\n\n\n    if not os.path.exists(filepath):\n        failed_files.append(filepath)\n        continue\n\n\n    chunks = load_and_chunk(filepath, rating=row.get('rating', None))\n\n    if not chunks:\n\n        failed_files.append(filepath)\n        continue\n\n\n    secondary_labels = parse_secondary_labels(row.get('secondary_labels', '[]'))\n\n\n    species_dir = os.path.join(CFG.SAVE_DIR, str(row['primary_label']))\n    os.makedirs(species_dir, exist_ok=True)\n\n\n    stem = Path(row['filename']).stem\n\n\n    spec0     = audio_to_melspec(chunks[0])\n    npy_path0 = os.path.join(species_dir, f'{stem}_chunk0.npy')\n    np.save(npy_path0, spec0)\n\n\n    df_all.at[idx, 'npy_path']               = npy_path0\n    df_all.at[idx, 'secondary_labels_parsed'] = str(secondary_labels)\n    df_all.at[idx, 'chunk_id']               = 0\n\n\n\n    if len(chunks) > 1:\n        spec1     = audio_to_melspec(chunks[1])\n        npy_path1 = os.path.join(species_dir, f'{stem}_chunk1.npy')\n        np.save(npy_path1, spec1)\n\n\n        new_row = row.to_dict()\n        new_row['npy_path']               = npy_path1\n        new_row['secondary_labels_parsed'] = str(secondary_labels)\n        new_row['chunk_id']               = 1\n        new_row['fold']                   = row['fold']  \n        extra_rows.append(new_row)\n\n\nprint(f'\\n{\"-\"*50}')\nprint(f'Processed        : {len(df_all) - len(failed_files):,}')\nprint(f'Failed/filtered  : {len(failed_files):,}')\nprint(f'Chunk-1 rows     : {len(extra_rows):,}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T12:37:04.665816Z","iopub.execute_input":"2026-05-01T12:37:04.666558Z","iopub.status.idle":"2026-05-01T13:56:09.749318Z","shell.execute_reply.started":"2026-05-01T12:37:04.666517Z","shell.execute_reply":"2026-05-01T13:56:09.746800Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# save final metadata CSV","metadata":{}},{"cell_type":"code","source":"\nif extra_rows:\n    df_chunks = pd.DataFrame(extra_rows)\n    df_all    = pd.concat([df_all, df_chunks], ignore_index=True)\n    print(f'Rows after chunk1 merge : {len(df_all):,}')\n\n\ndf_valid = df_all[df_all['npy_path'] != ''].reset_index(drop=True)\nprint(f'Valid rows (npy saved)  : {len(df_valid):,}')\nprint(f'Species                 : {df_valid[\"primary_label\"].nunique()}')\n\n\nprint(f'\\nFold distribution:')\nprint(df_valid['fold'].value_counts().sort_index().to_string())\n\n\nrequired_cols = [\n    'primary_label', 'secondary_labels_parsed', 'rating',\n    'filename', 'audio_dir', 'year', 'npy_path', 'chunk_id', 'fold'\n]\nprint(f'\\nColumn check:')\nall_ok = True\nfor col in required_cols:\n    ok = col in df_valid.columns\n    print(f'  {\"OK\" if ok else \"MISSING\":8s}  {col}')\n    if not ok:\n        all_ok = False\n\nif all_ok:\n    df_valid.to_csv(CFG.META_OUT, index=False)\n    print(f'\\nSaved: {CFG.META_OUT}')\nelse:\n    print('\\nFix missing columns before saving!')\n\n\nchunk0_only    = df_valid[df_valid['chunk_id'] == 0]\nspecies_counts = chunk0_only['primary_label'].value_counts()\nrare_mask      = species_counts < 30\nrare_species_list = species_counts[rare_mask].reset_index()\nrare_species_list.columns = ['primary_label', 'n_recordings']\n\n\nrare_with_info = rare_species_list.merge(\n    taxonomy[['primary_label', 'scientific_name', 'common_name', 'class_name']]\n    .assign(primary_label=taxonomy['inat_taxon_id'].astype(str)),\n    on='primary_label',\n    how='left'\n)\n\n\nrare_npy = (\n    chunk0_only[chunk0_only['primary_label'].isin(rare_species_list['primary_label'])]\n    [['primary_label', 'npy_path', 'filename', 'audio_dir', 'rating', 'year']]\n    .sort_values(['primary_label', 'rating'], ascending=[True, False])\n)\n\nRARE_SUMMARY_PATH = '/kaggle/working/rare_species_summary.csv'\nRARE_FILES_PATH   = '/kaggle/working/rare_species_all_files.csv'\n\nrare_with_info.sort_values('n_recordings').to_csv(RARE_SUMMARY_PATH, index=False)\nrare_npy.to_csv(RARE_FILES_PATH, index=False)\n\nprint(f'\\n=== Rare species (< 30 recordings) ===')\nprint(f'Count : {len(rare_with_info)} species out of {df_valid[\"primary_label\"].nunique()} total')\nprint(f'Saved : {RARE_SUMMARY_PATH}  (one row per species)')\nprint(f'Saved : {RARE_FILES_PATH}    (one row per file — use to locate audio)')\nprint()\nprint(rare_with_info.sort_values('n_recordings').to_string(index=False))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T14:28:24.406539Z","iopub.execute_input":"2026-05-01T14:28:24.411358Z","iopub.status.idle":"2026-05-01T14:28:26.355232Z","shell.execute_reply.started":"2026-05-01T14:28:24.411267Z","shell.execute_reply":"2026-05-01T14:28:26.353619Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_valid[required_cols].head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T14:28:26.357890Z","iopub.execute_input":"2026-05-01T14:28:26.358322Z","iopub.status.idle":"2026-05-01T14:28:26.390981Z","shell.execute_reply.started":"2026-05-01T14:28:26.358291Z","shell.execute_reply":"2026-05-01T14:28:26.389754Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Sample spectrograms","metadata":{}},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 3, figsize=(15, 7))\nsamples   = df_valid.sample(6, random_state=42)\n\nfor ax, (_, row) in zip(axes.flat, samples.iterrows()):\n    spec = np.load(row['npy_path'])\n    img  = ax.imshow(spec, aspect='auto', origin='lower',\n                     cmap='magma', vmin=0, vmax=1)\n    ax.set_title(\n        f\"{row['primary_label']}  fold={int(row['fold'])}  chunk={int(row['chunk_id'])}\\n\"\n        f\"shape={spec.shape}  year={int(row['year'])}\",\n        fontsize=8\n    )\n    ax.set_xlabel('Time frames')\n    ax.set_ylabel('Mel bins')\n    plt.colorbar(img, ax=ax, fraction=0.046)\n\nplt.suptitle('Sample spectrograms', fontsize=13, y=1.01)\nplt.tight_layout()\nplt.savefig('/kaggle/working/sample_spectrograms.png', dpi=100, bbox_inches='tight')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-01T14:28:26.392908Z","iopub.execute_input":"2026-05-01T14:28:26.393444Z","iopub.status.idle":"2026-05-01T14:28:30.556228Z","shell.execute_reply.started":"2026-05-01T14:28:26.393400Z","shell.execute_reply":"2026-05-01T14:28:30.554836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}