{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":227214445,"sourceType":"kernelVersion"},{"sourceId":229905901,"sourceType":"kernelVersion"},{"sourceId":230048145,"sourceType":"kernelVersion"},{"sourceId":236498053,"sourceType":"kernelVersion"},{"sourceId":236509965,"sourceType":"kernelVersion"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport plotly.express as px\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:15.193753Z","iopub.execute_input":"2025-05-02T04:39:15.194058Z","iopub.status.idle":"2025-05-02T04:39:16.723836Z","shell.execute_reply.started":"2025-05-02T04:39:15.194012Z","shell.execute_reply":"2025-05-02T04:39:16.722749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"taxonomy = pd.read_csv('/kaggle/input/birdclef-2025/taxonomy.csv')\ndf = pd.read_csv('/kaggle/input/birdclef-2025/train.csv').merge(\n    taxonomy[['primary_label', 'class_name']], \n    on='primary_label',\n    how='inner',\n    validate=\"m:1\",\n).merge(\n    pd.read_parquet('/kaggle/input/bc25-audio-stats/train_metadata_stats.parquet'), \n    on='filename',\n    how='inner',\n    validate=\"1:1\"\n).merge(\n    pd.read_parquet('/kaggle/input/bc25-spec-stats/quantize_params.parquet')\\\n    .rename(columns={'min_value': 'min_spec_value', 'max_value': 'max_spec_value'}),\n    on='filename',\n    how='inner',\n    validate=\"1:1\"\n).merge(\n    pd.read_parquet('/kaggle/input/bc25-folds/folds.parquet'),\n    on='filename',\n    how='inner',\n    validate=\"1:1\"\n).merge(\n    pd.read_parquet('/kaggle/input/analyze-voice-crops/train_voice_data.parquet'),\n    on='filename',\n    how='inner',\n    validate=\"1:1\"\n).merge(\n    pd.read_parquet('/kaggle/input/bc24-train-embedding-and-umap/train_embedding.parquet')[[\n        'filename',\n        'umap_component_1',\n        'umap_component_2',\n    ]],\n    on='filename',\n    how='inner',\n    validate=\"1:1\"\n)\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:16.724846Z","iopub.execute_input":"2025-05-02T04:39:16.725377Z","iopub.status.idle":"2025-05-02T04:39:23.009380Z","shell.execute_reply.started":"2025-05-02T04:39:16.725348Z","shell.execute_reply":"2025-05-02T04:39:23.008372Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"list(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:23.010293Z","iopub.execute_input":"2025-05-02T04:39:23.010541Z","iopub.status.idle":"2025-05-02T04:39:23.017065Z","shell.execute_reply.started":"2025-05-02T04:39:23.010520Z","shell.execute_reply":"2025-05-02T04:39:23.015951Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# File length statistics","metadata":{}},{"cell_type":"code","source":"(df.numel / 32000).describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:23.019402Z","iopub.execute_input":"2025-05-02T04:39:23.019667Z","iopub.status.idle":"2025-05-02T04:39:23.078608Z","shell.execute_reply.started":"2025-05-02T04:39:23.019645Z","shell.execute_reply":"2025-05-02T04:39:23.077577Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Classes","metadata":{}},{"cell_type":"code","source":"df.primary_label.nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:23.080512Z","iopub.execute_input":"2025-05-02T04:39:23.080863Z","iopub.status.idle":"2025-05-02T04:39:23.092266Z","shell.execute_reply.started":"2025-05-02T04:39:23.080835Z","shell.execute_reply":"2025-05-02T04:39:23.091342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"category_counts = df['class_name'].value_counts()\nprint('distinct classes:', len(category_counts))\nplt.figure(figsize=(8, 8))\nplt.pie(category_counts, labels=category_counts.index, autopct='%1.1f%%', startangle=140)\nplt.title('Class Distribution')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:23.093300Z","iopub.execute_input":"2025-05-02T04:39:23.093660Z","iopub.status.idle":"2025-05-02T04:39:23.338589Z","shell.execute_reply.started":"2025-05-02T04:39:23.093627Z","shell.execute_reply":"2025-05-02T04:39:23.337312Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = px.scatter_geo(df, lat='latitude', lon='longitude', color='class_name',\n                     hover_name='primary_label', projection=\"natural earth\",\n                     title='Locations by class')\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:23.339728Z","iopub.execute_input":"2025-05-02T04:39:23.340145Z","iopub.status.idle":"2025-05-02T04:39:25.299073Z","shell.execute_reply.started":"2025-05-02T04:39:23.340107Z","shell.execute_reply":"2025-05-02T04:39:25.298085Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[['class_name', 'min', 'mean', 'max']].groupby('class_name').agg('mean').style.bar(color='lightblue')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:25.300009Z","iopub.execute_input":"2025-05-02T04:39:25.300305Z","iopub.status.idle":"2025-05-02T04:39:25.413368Z","shell.execute_reply.started":"2025-05-02T04:39:25.300282Z","shell.execute_reply":"2025-05-02T04:39:25.412425Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Collections","metadata":{}},{"cell_type":"code","source":"category_counts = df['collection'].value_counts()\nplt.figure(figsize=(8, 8))\nplt.pie(category_counts, labels=category_counts.index, autopct='%1.1f%%', startangle=140)\nplt.title('Class Distribution')\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:25.414210Z","iopub.execute_input":"2025-05-02T04:39:25.414641Z","iopub.status.idle":"2025-05-02T04:39:25.532454Z","shell.execute_reply.started":"2025-05-02T04:39:25.414614Z","shell.execute_reply":"2025-05-02T04:39:25.531452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = px.scatter_geo(df, lat='latitude', lon='longitude', color='collection',\n                     hover_name='primary_label', projection=\"natural earth\",\n                     title='Locations by collection')\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:25.533434Z","iopub.execute_input":"2025-05-02T04:39:25.533763Z","iopub.status.idle":"2025-05-02T04:39:25.649985Z","shell.execute_reply.started":"2025-05-02T04:39:25.533737Z","shell.execute_reply":"2025-05-02T04:39:25.648829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"grouped = df.groupby(['class_name', 'collection']).size().unstack(fill_value=0)\ndisplay(grouped)\ngrouped.plot(kind='bar', stacked=True, figsize=(8, 6))\nplt.xlabel(\"class_name\")\nplt.ylabel(\"count\")\nplt.title(\"class_name and collection\")\nplt.legend(title=\"collection\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:25.651153Z","iopub.execute_input":"2025-05-02T04:39:25.651504Z","iopub.status.idle":"2025-05-02T04:39:26.039234Z","shell.execute_reply.started":"2025-05-02T04:39:25.651472Z","shell.execute_reply":"2025-05-02T04:39:26.038221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.groupby(['class_name', 'collection'])['std'].agg('mean')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:26.040228Z","iopub.execute_input":"2025-05-02T04:39:26.040568Z","iopub.status.idle":"2025-05-02T04:39:26.053549Z","shell.execute_reply.started":"2025-05-02T04:39:26.040541Z","shell.execute_reply":"2025-05-02T04:39:26.052575Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Secondary labels","metadata":{}},{"cell_type":"code","source":"has_secondary_labels = (df.secondary_labels != \"['']\") & (df.secondary_labels != \"[]\")\ndf.groupby(['class_name', 'collection', has_secondary_labels]).size().unstack(fill_value=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:26.056659Z","iopub.execute_input":"2025-05-02T04:39:26.057000Z","iopub.status.idle":"2025-05-02T04:39:26.083012Z","shell.execute_reply.started":"2025-05-02T04:39:26.056944Z","shell.execute_reply":"2025-05-02T04:39:26.082056Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"secondary_labels = (\n    df\n    .secondary_labels[has_secondary_labels]\n    .str.extractall(r\"'(?P<secondary_label>[^']+)'\")\n    .reset_index()\n    .merge(taxonomy[['primary_label', 'class_name']], left_on='secondary_label', right_on=\"primary_label\")\n    .drop(columns=[\"primary_label\", \"match\"])\n    .merge(df[['primary_label', 'class_name']].reset_index(), left_on='level_0', right_on=\"index\", suffixes=(\"\", \"_primary\"))\n    .drop(columns=[\"level_0\"])\n    .rename(columns={\"class_name\": \"secondary_class_name\", \"class_name_primary\": \"primary_class_name\"})\n)\nsecondary_labels.groupby(['primary_class_name', 'secondary_class_name']).size().unstack(fill_value=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:26.084551Z","iopub.execute_input":"2025-05-02T04:39:26.084916Z","iopub.status.idle":"2025-05-02T04:39:26.125283Z","shell.execute_reply.started":"2025-05-02T04:39:26.084881Z","shell.execute_reply":"2025-05-02T04:39:26.124307Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Human Voice","metadata":{}},{"cell_type":"code","source":"print('recordings with voice', df.has_voice.sum())\nprint()\n\nall_labels = df.primary_label.value_counts()\nfor collection in df.collection.unique():\n    df_csa = df[df.collection == collection]\n    print(f'% of {collection} recordings with voice:', round(sum(df_csa.voice_time > 0) / len(df_csa) * 100, 3))\n    print(f'% of {collection} recording samples with voice:', round(100 * df_csa.voice_time.sum() / (df_csa.numel.sum() / 32000), 3))\n    lost_recs = ((df_csa.numel / 32000 - df_csa.voice_time) < 5) * (df_csa.voice_time > 0)\n    print(f'# of {collection} recordings < 5 sec remaining:', sum(lost_recs))\n    lost_labels = df_csa[lost_recs].primary_label.value_counts()\n    num_lost = [label for label, c in lost_labels.items() if all_labels[label] - c <= 0]\n    print(f'# of lost primary labels:', num_lost) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:26.126562Z","iopub.execute_input":"2025-05-02T04:39:26.126841Z","iopub.status.idle":"2025-05-02T04:39:26.173688Z","shell.execute_reply.started":"2025-05-02T04:39:26.126816Z","shell.execute_reply":"2025-05-02T04:39:26.172695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def f(df, kind):\n    fig, ax1 = plt.subplots(figsize=(10, 6))\n    collections = df.index\n    x = range(len(collections))\n    ax1.bar(x, df['mean'] * 100, width=0.4, align='center', color=\"#1f77b4\")\n    ax1.set_ylabel(f'% of {kind} in collection have voice', color=\"#1f77b4\")\n    ax1.set_ylim(0, 100)\n    ax2 = ax1.twinx()\n    ax2.bar([i + 0.4 for i in x], df['sum'], width=0.4, align='center', color=\"#ff7f0e\")\n    ax2.set_ylabel(f'# of {kind} with voice', color=\"#ff7f0e\")\n    ax1.set_xticks([i + 0.2 for i in x])\n    ax1.set_xticklabels(collections)\n    fig.suptitle(f'{kind} with human voice per collection')\n    plt.show()\n    display(df)\n\nf(df.groupby(['collection']).has_voice.agg(['mean', 'sum']), 'recordings')\n\nvoice_samples = df.groupby('collection').voice_time.sum()\nf(pd.DataFrame(dict(\n    mean=voice_samples / (df.groupby(['collection']).numel.sum() / 32000),\n    sum=voice_samples, \n), index=df.collection.unique()), \"samples\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:26.174734Z","iopub.execute_input":"2025-05-02T04:39:26.175109Z","iopub.status.idle":"2025-05-02T04:39:26.622887Z","shell.execute_reply.started":"2025-05-02T04:39:26.175074Z","shell.execute_reply":"2025-05-02T04:39:26.622084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"voice_authors = df.groupby(['author', 'collection']).has_voice.agg(['mean', 'sum'])\n# take all authors that have more than 2 recordings and > 50% recordings with voice\nvoice_authors = voice_authors[(voice_authors['sum'] > 2) * (voice_authors['mean'] > .5)]\ndf['filter_voice'] = df.has_voice * (\n    (df.collection == 'CSA') | # all CSA recordings\n    df.author.isin({a for a, _ in voice_authors.index}) # filtered authors from the rest\n)\nvoice_authors","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:26.623800Z","iopub.execute_input":"2025-05-02T04:39:26.624119Z","iopub.status.idle":"2025-05-02T04:39:26.654038Z","shell.execute_reply.started":"2025-05-02T04:39:26.624096Z","shell.execute_reply":"2025-05-02T04:39:26.653055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.has_voice.agg(['mean', 'sum']), df[df.filter_voice].has_voice.agg(['mean', 'sum'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:26.654856Z","iopub.execute_input":"2025-05-02T04:39:26.655136Z","iopub.status.idle":"2025-05-02T04:39:26.666375Z","shell.execute_reply.started":"2025-05-02T04:39:26.655113Z","shell.execute_reply":"2025-05-02T04:39:26.665398Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"100 * df[df.filter_voice].voice_time.sum() / df.voice_time.sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:26.667265Z","iopub.execute_input":"2025-05-02T04:39:26.667587Z","iopub.status.idle":"2025-05-02T04:39:26.680552Z","shell.execute_reply.started":"2025-05-02T04:39:26.667552Z","shell.execute_reply":"2025-05-02T04:39:26.679534Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Embedding vizualization","metadata":{}},{"cell_type":"code","source":"!pip install -Uqq \"vegafusion[embed]>=1.5.0\"\n!pip install -Uqq \"vl-convert-python>=1.6.0\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:26.681530Z","iopub.execute_input":"2025-05-02T04:39:26.681847Z","iopub.status.idle":"2025-05-02T04:39:38.743299Z","shell.execute_reply.started":"2025-05-02T04:39:26.681815Z","shell.execute_reply":"2025-05-02T04:39:38.741641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import altair as alt\nalt.data_transformers.enable(\"vegafusion\");\n\ndef create_scatter_plot(df, color=\"common_name\", dot_size=50, title=\"Species Embeddings\"):\n    selection = alt.selection_point(fields=[color], bind='legend')\n    return alt.Chart(df).mark_circle(\n        size=dot_size,\n        opacity=0.7\n    ).encode(\n        x='umap_component_1',\n        y='umap_component_2',\n        color=alt.Color(color, legend=alt.Legend(\n            title='Species',\n            orient='right',\n            symbolLimit=50\n        )),\n        tooltip=[\n            alt.Tooltip('common_name', title='Common'),\n            alt.Tooltip('scientific_name', title='Scientific'),\n            alt.Tooltip('filename', title='File')\n        ],\n        opacity=alt.condition(selection, alt.value(1.0), alt.value(0.01))\n    ).add_params(\n        selection\n    ).properties(\n        width=600,\n        height=600,\n        title=title\n    ).interactive()\n\n\ncreate_scatter_plot(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:38.744512Z","iopub.execute_input":"2025-05-02T04:39:38.744796Z","iopub.status.idle":"2025-05-02T04:39:40.048599Z","shell.execute_reply.started":"2025-05-02T04:39:38.744771Z","shell.execute_reply":"2025-05-02T04:39:40.047431Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save everything","metadata":{}},{"cell_type":"code","source":"df.to_parquet('train_metadata_joined.parquet')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-02T04:39:40.049624Z","iopub.execute_input":"2025-05-02T04:39:40.050049Z","iopub.status.idle":"2025-05-02T04:39:40.210304Z","shell.execute_reply.started":"2025-05-02T04:39:40.050015Z","shell.execute_reply":"2025-05-02T04:39:40.209233Z"}},"outputs":[],"execution_count":null}]}