{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\nThis notebook intends to understand the data in the birdCLEF_2025 dataset and create some simple classifire models for recognizing species by ther soud.","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport librosa\nfrom IPython.display import Audio\nfrom sklearn.metrics import roc_curve","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T13:30:58.095752Z","iopub.execute_input":"2025-05-24T13:30:58.096059Z","iopub.status.idle":"2025-05-24T13:30:58.101080Z","shell.execute_reply.started":"2025-05-24T13:30:58.096040Z","shell.execute_reply":"2025-05-24T13:30:58.099966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data\nINPUT_PATH = '/kaggle/input/birdclef-2025/'\nTRAIN_AUDIO_PATH = os.path.join(INPUT_PATH, 'train_audio')\nTEST_SOUNDSCAPES_PATH = os.path.join(INPUT_PATH, 'test_soundscapes')\nTRAIN_SOUNDSCAPES_PATH = os.path.join(INPUT_PATH, 'train_soundscapes')\n\n# Load data\ntaxonomy = pd.read_csv(os.path.join(INPUT_PATH, 'taxonomy.csv'))\ntrain_meta = pd.read_csv(os.path.join(INPUT_PATH, 'train.csv'))\n# recording_locations = pd.read_csv(os.path.join(INPUT_PATH, 'recording_location.txt'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T13:30:58.102909Z","iopub.execute_input":"2025-05-24T13:30:58.103296Z","iopub.status.idle":"2025-05-24T13:30:58.241473Z","shell.execute_reply.started":"2025-05-24T13:30:58.103269Z","shell.execute_reply":"2025-05-24T13:30:58.240428Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Understanding our data","metadata":{}},{"cell_type":"code","source":"train_meta.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T13:30:58.242871Z","iopub.execute_input":"2025-05-24T13:30:58.243230Z","iopub.status.idle":"2025-05-24T13:30:58.259943Z","shell.execute_reply.started":"2025-05-24T13:30:58.243202Z","shell.execute_reply":"2025-05-24T13:30:58.258688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# link the filenames to our dataset\ndef preprocess_train_meta(df):\n    \"\"\"Preprocesses the training metadata.\"\"\"\n    df['secondary_labels'] = df['secondary_labels'].apply(lambda x: re.findall(r\"'(\\w+)'\", x))\n    df['file_path'] = df.apply(lambda row: os.path.join(TRAIN_AUDIO_PATH, row['filename']), axis=1)\n    return df\n\ntrain_meta = preprocess_train_meta(train_meta)\ntrain_meta.dtypes\ntrain_meta.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T13:30:58.260997Z","iopub.execute_input":"2025-05-24T13:30:58.261326Z","iopub.status.idle":"2025-05-24T13:30:58.492028Z","shell.execute_reply.started":"2025-05-24T13:30:58.261297Z","shell.execute_reply":"2025-05-24T13:30:58.490822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"taxonomy.head() # we see that taxonomy allow us to match 'primarry_lables' of train.csv to the species\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T13:30:58.493747Z","iopub.execute_input":"2025-05-24T13:30:58.494188Z","iopub.status.idle":"2025-05-24T13:30:58.504462Z","shell.execute_reply.started":"2025-05-24T13:30:58.494155Z","shell.execute_reply":"2025-05-24T13:30:58.503372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nplt.figure(figsize=(14, 6))\n\nplt.subplot(1, 2, 1)\nsns.countplot(data=taxonomy, y='class_name', order=taxonomy['class_name'].value_counts().index)\nplt.title('Distribution of Class Names')\nplt.xlabel('Count')\nplt.ylabel('Class Name')\n\nplt.subplot(1, 2, 2)\nsns.countplot(data=taxonomy, y='common_name', order=taxonomy['common_name'].value_counts().index)\nplt.title('Distribution of Common Names')\nplt.xlabel('Count')\nplt.ylabel('Common Name')\n\nplt.tight_layout()\nplt.show()\n\n# Check if the two columns are identical\nidentical = taxonomy['inat_taxon_id'].equals(taxonomy['primary_label'])\nprint(f\"Are 'class_name' and 'common_name' identical? {identical}\")\n\n# Show rows where they differ, if any\ndiff_rows = taxonomy[taxonomy['class_name'] != taxonomy['common_name']]\ndiff_rows","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T13:30:58.505475Z","iopub.execute_input":"2025-05-24T13:30:58.505730Z","iopub.status.idle":"2025-05-24T13:31:00.788528Z","shell.execute_reply.started":"2025-05-24T13:30:58.505710Z","shell.execute_reply":"2025-05-24T13:31:00.787556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# lets explore an audio file\naudio_examples = train_meta.sample(1)\nfile_path = audio_examples['file_path'].tolist()[0]\ntitle = audio_examples['primary_label'].tolist()[0]\n\n\nAudio(file_path)\ny, sr = librosa.load(file_path)\n# Plot waveform\nplt.figure(figsize=(14, 4))\nplt.subplot(1, 2, 1)\nlibrosa.display.waveshow(y, sr=sr)\nplt.title(\"Waveform: \" + title )\n\n# Plot spectrogram\nplt.subplot(1, 2, 2)\nD = librosa.stft(y)\nS_db = librosa.amplitude_to_db(np.abs(D), ref=np.max)\nlibrosa.display.specshow(S_db, sr=sr, x_axis='time', y_axis='log')\nplt.title(f'Spectrogram: {title}')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T13:31:00.789775Z","iopub.execute_input":"2025-05-24T13:31:00.790044Z","iopub.status.idle":"2025-05-24T13:31:01.732930Z","shell.execute_reply.started":"2025-05-24T13:31:00.790023Z","shell.execute_reply":"2025-05-24T13:31:01.731307Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Feature extraction\nbased on [1] and [2] we decidet to use Mel-Frequency Cepstral Coefficients (MFCCs) to train our model","metadata":{}},{"cell_type":"code","source":"# Feature Extraction (Simple MFCC)\ndef extract_mfcc(file_path, sr=22050, n_mfcc=20):\n    \"\"\"Extracts MFCC features from an audio file.\"\"\"\n    try:\n        y, sr = librosa.load(file_path, sr=sr)\n        mfccs = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=n_mfcc)\n        mfccs_processed = np.mean(mfccs.T, axis=0)  # Average across time\n    except Exception as e:\n        # Comment out this line to suppress error messages\n        # print(f\"Error processing {file_path}: {e}\")\n        return None\n    return mfccs_processed\n\n# Example MFCC extraction\nexample_file = train_meta['file_path'].iloc[0]\nmfccs = extract_mfcc(example_file)\nprint(\"\\nMFCC Features Example:\", mfccs)\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T13:35:47.002782Z","iopub.execute_input":"2025-05-24T13:35:47.003187Z","iopub.status.idle":"2025-05-24T13:35:47.380188Z","shell.execute_reply.started":"2025-05-24T13:35:47.003157Z","shell.execute_reply":"2025-05-24T13:35:47.378977Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train Models\n- SVM\n- gradiant boosting\n- ","metadata":{}},{"cell_type":"code","source":"# Model Training (XGBoost or LightGBM)\ndef train_model(train_meta, model_type='xgboost', n_samples=500, n_mfcc=20):\n    \"\"\"Trains a XGBoost or LightGBM model.\"\"\"\n    # Sample a subset of data for faster training\n    train_subset = train_meta.sample(n_samples, random_state=42)\n\n    # Extract MFCC features\n    features = []\n    labels = []\n    for index, row in train_subset.iterrows():\n        mfccs = extract_mfcc(row['file_path'], n_mfcc=n_mfcc)\n        if mfccs is not None:\n            features.append(mfccs)\n            labels.append(row['primary_label'])\n\n    X = np.array(features)\n    y = np.array(labels)\n\n    # Train/Test split\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n\n    if model_type == 'xgboost':\n        model = xgb.XGBClassifier(use_label_encoder=False, eval_metric='logloss', random_state=42) #added eval_metric as it is needed\n        model.fit(X_train, y_train)\n\n    elif model_type == 'lightgbm':\n        model = lgb.LGBMClassifier(random_state=42)\n        model.fit(X_train, y_train)\n\n    else:\n        raise ValueError(\"Invalid model_type. Choose 'xgboost' or 'lightgbm'.\")\n\n    return model, X_test, y_test, y_train\n\n#Choose between 'xgboost' or 'lightgbm'\nmodel, X_test, y_test, y_train = train_model(train_meta, model_type='lightgbm') #or 'xgboost'","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model evaluations\n\nfpr, tpr, thresholds = roc_curve(y_true, y_score)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Sources:\n- https://www.kaggle.com/code/jocelyndumlao/birdclef-2025-mfcc-feature-roc-auc-analysis#Import-Libraries (24.05.2025)\n- https://github.com/UtrechtUniversity/animal-sounds (24.05.2025)","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}