{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":70203,"databundleVersionId":8068726},{"sourceType":"modelInstanceVersion","sourceId":516989,"databundleVersionId":13353982,"modelInstanceId":404337,"modelId":319}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BirdCLEF 2024: Full 182-Species Embedding Extraction\nThis notebook extracts Perch v2 embeddings for all 182 species in the competition.\nEstimated runtime: ~45-60 minutes on T4 GPU.","metadata":{}},{"cell_type":"code","source":"# Mandatory setup for Perch and BirdNET to work correctly on Kaggle\n!pip install -q --upgrade \"tensorflow[and-cuda]>=2.16.1\" birdnetlib pyarrow scikit-learn resampy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T19:30:30.719239Z","iopub.execute_input":"2026-04-26T19:30:30.719950Z","iopub.status.idle":"2026-04-26T19:32:12.474033Z","shell.execute_reply.started":"2026-04-26T19:30:30.719913Z","shell.execute_reply":"2026-04-26T19:32:12.473135Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nprint(\"Num GPUs Available: \", len(tf.config.list_physical_devices('GPU')))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T19:32:12.476086Z","iopub.execute_input":"2026-04-26T19:32:12.476624Z","iopub.status.idle":"2026-04-26T19:32:35.001022Z","shell.execute_reply.started":"2026-04-26T19:32:12.476568Z","shell.execute_reply":"2026-04-26T19:32:35.000154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport librosa\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\n\n# --- Configuration ---\nINPUT_DIR = Path('/kaggle/input')\nAUDIO_DIR = INPUT_DIR / 'competitions/birdclef-2024/train_audio'\nOUTPUT_DIR = Path('/kaggle/working/cache/embeddings')\nOUTPUT_DIR.mkdir(parents=True, exist_ok=True)\nSR = 32000\nCHUNK_SEC = 5.0\n\n# --- Model Loading ---\nimport tensorflow as tf\nfrom pathlib import Path\nimport numpy as np\nimport librosa\n\nimport tensorflow as tf\nimport numpy as np\n\nMODEL_PATH = \"/kaggle/input/models/google/bird-vocalization-classifier/tensorflow2/perch_v2/2\"\n\ntry:\n    print(f\"Loading Perch v2...\")\n    model = tf.saved_model.load(MODEL_PATH)\n    \n    # Perch v2 on Kaggle usually uses 'serving_default' as the signature name\n    if 'serving_default' in model.signatures:\n        perch_fn = model.signatures['serving_default']\n        print(\"Success: Using 'serving_default' signature.\")\n    else:\n        # Fallback: Use the first available signature\n        first_sig = list(model.signatures.keys())[0]\n        perch_fn = model.signatures[first_sig]\n        print(f\"Success: Using '{first_sig}' signature.\")\n        \nexcept Exception as e:\n    print(f\"Failed to load model: {e}\")\n    perch_fn = None\n\ndef extract_perch(audio_5s_32k):\n    if perch_fn is None: raise RuntimeError(\"Model not loaded.\")\n    \n    # Input must be [batch, samples]\n    x = tf.constant(audio_5s_32k[None, :], dtype=tf.float32)\n    \n    # Signature calls usually require 'inputs=' or 'input_node='\n    # Most Perch models use 'inputs'\n    try:\n        outputs = perch_fn(inputs=x)\n    except TypeError:\n        outputs = perch_fn(x)\n        \n    # Perch v2 returns a dict with 'embedding' and 'label'\n    return outputs['embedding'].numpy()[0]\n\n# --- Metadata Setup ---\ntrain_df = pd.read_csv(INPUT_DIR / 'competitions/birdclef-2024/train_metadata.csv')\n# Subsample to 100 files per species for a balanced but fast run\nsampled_df = train_df.groupby('primary_label').head(100).reset_index(drop=True)\nprint(f\"Processing {len(sampled_df)} files across {train_df['primary_label'].nunique()} species.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T19:33:42.219742Z","iopub.execute_input":"2026-04-26T19:33:42.220245Z","iopub.status.idle":"2026-04-26T19:33:44.486321Z","shell.execute_reply.started":"2026-04-26T19:33:42.220211Z","shell.execute_reply":"2026-04-26T19:33:44.485198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm.notebook import tqdm\n# 1. Verify there is data\nprint(f\"Total files to process: {len(sampled_df)}\")\nresults = []\nfor idx, row in tqdm(sampled_df.iterrows(), total=len(sampled_df)):\n    path = AUDIO_DIR / row['filename']\n    if not path.exists(): continue\n    \n    try:\n        # Load only the first 5 seconds to be lightning fast\n        y, _ = librosa.load(path, sr=SR, duration=CHUNK_SEC, mono=True)\n        if len(y) < SR * CHUNK_SEC:\n            y = np.pad(y, (0, int(SR * CHUNK_SEC) - len(y)))\n        \n        emb = extract_perch(y)\n        results.append({\n            'file_id': row['filename'],\n            'species_code': row['primary_label'],\n            'emb': emb\n        })\n        \n        # Checkpoint every 1000 files\n        if (idx + 1) % 1000 == 0:\n            pd.DataFrame(results).to_parquet(OUTPUT_DIR / f'perch_train_182_{idx}.parquet')\n            results = []\n    except Exception as e:\n        print(f\"Error processing {row['filename']}: {e}\")\n\n# Final Save\nif results:\n    pd.DataFrame(results).to_parquet(OUTPUT_DIR / 'perch_train_182_final.parquet')\nprint(\"Extraction Complete!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-04-26T19:38:43.174432Z","iopub.execute_input":"2026-04-26T19:38:43.174977Z","execution_failed":"2026-04-26T19:39:01.319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}