{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-19T00:48:39.221105Z","iopub.execute_input":"2025-03-19T00:48:39.221360Z","iopub.status.idle":"2025-03-19T00:49:31.888314Z","shell.execute_reply.started":"2025-03-19T00:48:39.221331Z","shell.execute_reply":"2025-03-19T00:49:31.887439Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install librosa --quiet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:02:09.654386Z","iopub.execute_input":"2025-03-19T01:02:09.654738Z","iopub.status.idle":"2025-03-19T01:02:14.772202Z","shell.execute_reply.started":"2025-03-19T01:02:09.654708Z","shell.execute_reply":"2025-03-19T01:02:14.770915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport re\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\nimport soundfile as sf\nimport plotly.graph_objects as go\nimport torch\nimport torchaudio\nimport requests\nimport xgboost as xgb\nimport lightgbm as lgb\n\nfrom sklearn.metrics import roc_auc_score, roc_curve\nfrom sklearn.preprocessing import MultiLabelBinarizer\nfrom sklearn.model_selection import train_test_split\nfrom urllib.request import urlopen\nfrom datetime import datetime, timedelta\nfrom scipy.interpolate import interp1d\nfrom bs4 import BeautifulSoup as bs\nfrom tqdm.notebook import tqdm\nfrom PIL import Image\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:02:17.764287Z","iopub.execute_input":"2025-03-19T01:02:17.764648Z","iopub.status.idle":"2025-03-19T01:02:17.771193Z","shell.execute_reply.started":"2025-03-19T01:02:17.764623Z","shell.execute_reply":"2025-03-19T01:02:17.770276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define dataset paths\nINPUT_PATH = '/kaggle/input/birdclef-2025/'\nTRAIN_AUDIO_PATH = os.path.join(INPUT_PATH, 'train_audio')\nTEST_SOUNDSCAPES_PATH = os.path.join(INPUT_PATH, 'test_soundscapes')\n\n# Load necessary data files\nbiodiversity_df = pd.read_csv(os.path.join(INPUT_PATH, 'taxonomy.csv'))\ntrain = pd.read_csv(os.path.join(INPUT_PATH, 'train.csv'))\nsample_submission_df = pd.read_csv(os.path.join(INPUT_PATH, 'sample_submission.csv'))\nrecording_locations_df = pd.read_csv(os.path.join(INPUT_PATH, 'recording_location.txt'))\n\n# Process secondary_labels in training data\ntrain = pd.read_csv(os.path.join(INPUT_PATH, 'train.csv'))\ntrain['secondary_labels'] = train['secondary_labels'].apply(lambda x: re.findall(r\"'(\\w+)'\", x))\n\n# Add column for count of secondary labels\ntrain['secondary_labels_count'] = train['secondary_labels'].apply(len)\n\n# Display the first few rows of training data\ntrain.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:02:20.702355Z","iopub.execute_input":"2025-03-19T01:02:20.702677Z","iopub.status.idle":"2025-03-19T01:02:20.971351Z","shell.execute_reply.started":"2025-03-19T01:02:20.702654Z","shell.execute_reply":"2025-03-19T01:02:20.970581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prepare data for visualization\ngrouped_df = train.groupby(['primary_label', 'latitude', 'longitude']).size().reset_index(name='count')\ntrain_plot_df = train.merge(grouped_df, on=['primary_label', 'latitude', 'longitude'], how='left').dropna(subset=['count'])\ntrain_plot_df['count'] = train_plot_df['count'].astype(int)\n\n# Radius scaling interpolation\ncounts = train_plot_df['count'].tolist()\nradius_scale = interp1d([min(counts), max(counts)], [3, 20])\nradius_values = radius_scale(counts)\n\n# Create interactive map visualization\nfig = go.Figure(go.Densitymapbox(\n    lat=train_plot_df['latitude'],\n    lon=train_plot_df['longitude'],\n    z=train_plot_df['count'],\n    radius=radius_values,\n    colorscale=\"Rainbow\",\n    opacity=0.7,\n    colorbar=dict(title=\"Observation Count\")\n))\n\n# Map layout adjustments\nfig.update_layout(\n    title=\"Geographic Distribution of Bird Observations\",\n    title_x=0.5,\n    mapbox_style=\"carto-positron\",\n    height=800,\n    mapbox=dict(\n        zoom=2,\n        center=dict(lat=train_plot_df['latitude'].mean(), lon=train_plot_df['longitude'].mean()),\n    ),\n    margin=dict(r=0, l=0, t=50, b=0),\n)\n\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:02:24.795939Z","iopub.execute_input":"2025-03-19T01:02:24.796238Z","iopub.status.idle":"2025-03-19T01:02:24.868536Z","shell.execute_reply.started":"2025-03-19T01:02:24.796217Z","shell.execute_reply":"2025-03-19T01:02:24.867721Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set up an optimized figure layout for data visualizations\nfig, axes = plt.subplots(2, 2, figsize=(14, 10))\n\n# Histogram for Latitude distribution\ntrain['latitude'].hist(bins=50, color='skyblue', ax=axes[0, 0])\naxes[0, 0].set_title('Latitude Distribution', fontsize=14)\naxes[0, 0].set_xlabel('Latitude', fontsize=12)\naxes[0, 0].set_ylabel('Count', fontsize=12)\n\n# Histogram for Longitude distribution\ntrain['longitude'].hist(bins=50, color='lightgreen', ax=axes[0, 1])\naxes[0, 1].set_title('Longitude Distribution', fontsize=14)\naxes[0, 1].set_xlabel('Longitude', fontsize=12)\naxes[0, 1].set_ylabel('Count', fontsize=12)\n\n# Scatter plot showing geographic distribution of recordings\ntrain.plot.scatter(x='longitude', y='latitude', alpha=0.3, color='coral', ax=axes[1, 0])\naxes[1, 0].set_title('Geographic Distribution of Recordings', fontsize=14)\naxes[1, 0].set_xlabel('Longitude', fontsize=12)\naxes[1, 0].set_ylabel('Latitude', fontsize=12)\n\n# Horizontal bar chart of top 10 authors by number of recordings\ntrain['author'].value_counts().nlargest(10).plot.barh(color='orchid', ax=axes[1, 1])\naxes[1, 1].set_title('Top 10 Authors by Recordings', fontsize=14)\naxes[1, 1].set_xlabel('Number of Recordings', fontsize=12)\naxes[1, 1].set_ylabel('Author', fontsize=12)\n\n# Adjust layout to prevent overlap and improve readability\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:02:32.429524Z","iopub.execute_input":"2025-03-19T01:02:32.429860Z","iopub.status.idle":"2025-03-19T01:02:33.413408Z","shell.execute_reply.started":"2025-03-19T01:02:32.429837Z","shell.execute_reply":"2025-03-19T01:02:33.412604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load biodiversity taxonomy data\nbiodiversity_df = pd.read_csv(\"/kaggle/input/birdclef-2025/taxonomy.csv\")\n\n# Set pandas to display all columns\npd.set_option('display.max_columns', None)\n\n# Display initial rows of the dataset\ndisplay(biodiversity_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:02:36.693059Z","iopub.execute_input":"2025-03-19T01:02:36.693348Z","iopub.status.idle":"2025-03-19T01:02:36.704934Z","shell.execute_reply.started":"2025-03-19T01:02:36.693327Z","shell.execute_reply":"2025-03-19T01:02:36.704205Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Visualize top 20 animal classes\nclass_counts = biodiversity_df['class_name'].value_counts().head(20)\n\nplt.figure(figsize=(16, 8))\nclass_counts.plot.barh(color='forestgreen')\n\n# Customize plot aesthetics\nplt.title('Top 20 Animal Classes in Magdalena Valley', fontsize=18, color='darkorange')\nplt.xlabel('Occurrences', fontsize=14)\nplt.ylabel('Animal Class', fontsize=14)\n\n# Invert y-axis for readability\nplt.gca().invert_yaxis()\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:02:39.109100Z","iopub.execute_input":"2025-03-19T01:02:39.109386Z","iopub.status.idle":"2025-03-19T01:02:39.315000Z","shell.execute_reply.started":"2025-03-19T01:02:39.109364Z","shell.execute_reply":"2025-03-19T01:02:39.314194Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group biodiversity data by scientific and class names\nanimal_classification = biodiversity_df.groupby(['scientific_name', 'class_name']).size().reset_index(name='total_count')\n\n# Calculate proportions of each animal class\nclass_counts = biodiversity_df['class_name'].value_counts()\nclass_proportions = class_counts / class_counts.sum()\n\n# Plot distribution of animal classes\nplt.figure(figsize=(12, 8))\nplt.barh(class_proportions.index, class_proportions.values, color=plt.cm.coolwarm(np.linspace(0, 1, len(class_proportions))))\n\n# Enhance plot aesthetics\nplt.title('Distribution of Animal Classes in Magdalena Valley', fontsize=16)\nplt.xlabel('Proportion of Total Observations', fontsize=12)\nplt.ylabel('Animal Class', fontsize=12)\n\n# Annotate each bar with percentage values\nfor i, (value, label) in enumerate(zip(class_proportions.values, class_proportions.index)):\n    plt.text(value, i, f'{value:.1%}', va='center', ha='left', fontsize=10)\n\nplt.gca().invert_yaxis()  # Highest proportions at the top\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:02:42.822742Z","iopub.execute_input":"2025-03-19T01:02:42.823059Z","iopub.status.idle":"2025-03-19T01:02:43.019433Z","shell.execute_reply.started":"2025-03-19T01:02:42.823036Z","shell.execute_reply":"2025-03-19T01:02:43.018744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"TRAIN_AUDIO_PATH = '/kaggle/input/birdclef-2025/train_audio/'\ntrain['file_path'] = train['filename'].apply(\n    lambda x: os.path.join(TRAIN_AUDIO_PATH, x)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:02:45.869838Z","iopub.execute_input":"2025-03-19T01:02:45.870170Z","iopub.status.idle":"2025-03-19T01:02:45.903173Z","shell.execute_reply.started":"2025-03-19T01:02:45.870142Z","shell.execute_reply":"2025-03-19T01:02:45.902410Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import librosa\nimport numpy as np\n\ndef extract_audio_features(file_paths, n_mfcc=20, max_duration=5, sr=32000):\n    \"\"\"\n    Extract MFCC features from a list of audio file paths.\n    \n    Args:\n        file_paths (list or pd.Series): Paths to audio files.\n        n_mfcc (int): Number of MFCC features to extract.\n        max_duration (int): Maximum duration (in seconds) of audio to process.\n        sr (int): Sampling rate for audio loading.\n        \n    Returns:\n        np.ndarray: 2D array of shape (num_files, num_features)\n    \"\"\"\n    features = []\n    \n    for fp in file_paths:\n        # Load audio file with librosa, limit duration to avoid memory issues\n        audio, _ = librosa.load(fp, sr=sr, duration=max_duration)\n\n        # Ensure consistent length by padding or trimming\n        required_length = sr * max_duration\n        if len(audio) < required_length:\n            audio = np.pad(audio, (0, required_length - len(audio)))\n        else:\n            audio = audio[:required_length]\n\n        # Extract MFCC features\n        mfcc = librosa.feature.mfcc(y=audio, sr=sr, n_mfcc=n_mfcc)\n        \n        # Compute statistics over MFCC features (mean, std)\n        mfcc_mean = np.mean(mfcc, axis=1)\n        mfcc_std = np.std(mfcc, axis=1)\n\n        # Concatenate statistics into a single feature vector\n        feature_vector = np.concatenate([mfcc_mean, mfcc_std])\n\n        features.append(feature_vector)\n    \n    return np.array(features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:03:20.933440Z","iopub.execute_input":"2025-03-19T01:03:20.933756Z","iopub.status.idle":"2025-03-19T01:03:20.939834Z","shell.execute_reply.started":"2025-03-19T01:03:20.933730Z","shell.execute_reply":"2025-03-19T01:03:20.938872Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Feature extraction using MFCC\ndef extract_mfcc_features(audio_path, sample_rate=22050, n_mfcc=20):\n    \"\"\"Extracts MFCC features from an audio file.\"\"\"\n    try:\n        audio, sr = librosa.load(audio_path, sr=sample_rate)\n        mfcc = librosa.feature.mfcc(y=audio, sr=sr, n_mfcc=n_mfcc)\n        mfcc_mean = mfcc.mean(axis=1)\n        return mfcc_mean\n    except Exception:\n        return None\n\n# Model training function using XGBoost or LightGBM\ndef train_audio_classifier(data, model_type='lightgbm', sample_size=500, n_mfcc=20):\n    \"\"\"Trains an audio classifier using XGBoost or LightGBM with MFCC features.\"\"\"\n    data_sample = data.sample(sample_size, random_state=42)\n    \n    # Extract features\n    X, y = [], []\n    for _, row in data_sample.iterrows():\n        features = extract_mfcc_features(row['file_path'], n_mfcc=n_mfcc)\n        if features is not None:\n            X.append(features)\n            y.append(row['primary_label'])\n\n    X = np.array(X)\n    y = np.array(y)\n\n    # Train-test split\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n    if model_type == 'xgboost':\n        model = xgb.XGBClassifier(random_state=42, use_label_encoder=False)\n    elif model_type == 'lightgbm':\n        model = lgb.LGBMClassifier(random_state=42)\n    else:\n        raise ValueError(\"Model type must be either 'xgboost' or 'lightgbm'\")\n\n    model.fit(X_train, y_train)\n\n    return model, X_test, y_test\n\n# Example usage\nsample_audio_path = train.iloc[0]['file_path']\nsample_mfcc = extract_mfcc_features(sample_audio_path)\n\nprint(\"Extracted MFCC Features:\", sample_mfcc)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:03:23.897143Z","iopub.execute_input":"2025-03-19T01:03:23.897446Z","iopub.status.idle":"2025-03-19T01:03:24.168998Z","shell.execute_reply.started":"2025-03-19T01:03:23.897425Z","shell.execute_reply":"2025-03-19T01:03:24.168111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate_model(model, X_test, y_test, labels, y_train):\n    \"\"\"Evaluates the model performance using macro-averaged ROC-AUC.\"\"\"\n    y_pred_proba = model.predict_proba(X_test)\n\n    # Binarize labels for multi-class ROC evaluation\n    mlb = MultiLabelBinarizer(classes=labels)\n    mlb.fit([[lbl] for lbl in np.concatenate([y_train, y_test])])\n    y_test_bin = mlb.transform([[lbl] for lbl in y_test])\n\n    roc_auc_scores = []\n    fprs, tprs = [], []\n\n    for idx, label in enumerate(labels):\n        try:\n            label_idx = list(mlb.classes_).index(label)\n\n            # Check if ROC AUC can be calculated\n            if len(np.unique(y_test_bin[:, label_idx])) <= 1:\n                raise ValueError(f\"Only one class present in y_true for {label}.\")\n\n            roc_auc = roc_auc_score(y_test_bin[:, label_idx], y_pred_proba[:, label_idx])\n            fpr, tpr, _ = roc_curve(y_test_bin[:, label_idx], y_pred_proba[:, label_idx])\n\n            roc_auc_scores.append(roc_auc)\n            fprs.append(fpr)\n            tprs.append(tpr)\n\n        except (ValueError, IndexError) as e:\n            print(f\"Skipped label {label}: {e}\")\n            roc_auc_scores.append(np.nan)\n            fprs.append(None)\n            tprs.append(None)\n\n    # Compute macro-averaged ROC-AUC\n    macro_roc_auc = np.nanmean(roc_auc_scores)\n    print(f\"Macro-Averaged ROC AUC: {macro_roc_auc:.4f}\")\n\n    # Plot ROC curves\n    plt.figure(figsize=(10, 8))\n    for idx, label in enumerate(labels):\n        if fprs[idx] is not None and tprs[idx] is not None:\n            plt.plot(fprs[idx], tprs[idx], label=f'{label} (AUC: {roc_auc_scores[idx]:.2f})')\n\n    plt.plot([0, 1], [0, 1], 'k--', label='Random guess')\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title('ROC Curves by Class')\n    plt.legend(loc='lower right')\n    plt.grid(True)\n    plt.show()\n\n    return macro_roc_auc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:03:26.784802Z","iopub.execute_input":"2025-03-19T01:03:26.785121Z","iopub.status.idle":"2025-03-19T01:03:26.793600Z","shell.execute_reply.started":"2025-03-19T01:03:26.785093Z","shell.execute_reply":"2025-03-19T01:03:26.792604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_audio_classifier(train_df, model_type='lightgbm'):\n    # Example placeholder feature extraction\n    X = extract_audio_features(train_df['file_path'])\n    y = train_df['primary_label']\n\n    X_train, X_test, y_train, y_test = train_test_split(\n        X, y, test_size=0.2, random_state=42, stratify=y\n    )\n\n    if model_type == 'lightgbm':\n        model = lgb.LGBMClassifier(objective='multiclass', random_state=42)\n        model.fit(X_train, y_train)\n    else:\n        raise ValueError(\"Currently only 'lightgbm' is supported\")\n\n    return model, X_test, y_test, y_train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:03:29.144077Z","iopub.execute_input":"2025-03-19T01:03:29.144398Z","iopub.status.idle":"2025-03-19T01:03:29.149232Z","shell.execute_reply.started":"2025-03-19T01:03:29.144370Z","shell.execute_reply":"2025-03-19T01:03:29.148479Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train your model\nmodel, X_test, y_test, y_train = train_audio_classifier(train, model_type='lightgbm')\n\n# Extract labels for evaluation\nlabels = train['primary_label'].unique()\n\n# Evaluate your trained model\nmacro_roc_auc = evaluate_model(model, X_test, y_test, labels, y_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:03:30.992448Z","iopub.execute_input":"2025-03-19T01:03:30.992794Z","iopub.status.idle":"2025-03-19T01:17:12.631522Z","shell.execute_reply.started":"2025-03-19T01:03:30.992767Z","shell.execute_reply":"2025-03-19T01:17:12.630629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Prediction and Submission Function\ndef predict_and_submit(model, sample_submission_df, train, labels, n_mfcc=20):\n    predictions = {}\n\n    # Map file_id to file_path for efficiency\n    file_id_to_path = {\n        row['filename'].replace('.ogg', ''): row['file_path']\n        for _, row in train.iterrows()\n    }\n\n    for _, row in sample_submission_df.iterrows():\n        row_id = row['row_id']\n        file_id = row_id.split('_')[1]\n        audio_path = os.path.join(TEST_SOUNDSCAPES_PATH, f\"{file_id}.ogg\")\n\n        try:\n            mfccs = extract_mfcc_features(audio_path, n_mfcc=n_mfcc)\n\n            if mfccs is None:\n                prediction_probs = [0.01] * len(labels)\n            else:\n                prediction_probs = model.predict_proba([mfccs])[0]\n\n            label_predictions = dict(zip(labels, prediction_probs))\n\n        except (FileNotFoundError, KeyError) as e:\n            print(f\"Warning: {e}, assigning default probabilities for {row_id}.\")\n            label_predictions = {label: 0.01 for label in labels}\n\n        predictions[row_id] = label_predictions\n        \n    # Create and save submission DataFrame\n    submission_df = pd.DataFrame.from_dict(predictions, orient='index')\n    submission_df.index.name = 'row_id'\n    submission_df.reset_index(inplace=True)\n\n    submission_df.to_csv('submission.csv', index=False)\n    print(\"Submission file created successfully!\")\n\n    return submission_df\n\n# Generate the submission\nsubmission_df = predict_and_submit(model, sample_submission_df, train, labels)\n\n# Display submission file head\nprint(\"Submission File Preview:\")\nprint(submission_df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-19T01:26:37.988594Z","iopub.execute_input":"2025-03-19T01:26:37.988932Z","iopub.status.idle":"2025-03-19T01:26:39.254061Z","shell.execute_reply.started":"2025-03-19T01:26:37.988907Z","shell.execute_reply":"2025-03-19T01:26:39.253128Z"}},"outputs":[],"execution_count":null}]}