{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What the notebook does:\n\n* in-depth EDA for the scored bird classes with two additional datasets:\n    - 2020 IUCN Red List category\n    - The birds of the Hawaiian islands (Pyle's checklist)\n* visualize a few samples of all scored classes with mel-scaled spectrograms","metadata":{}},{"cell_type":"code","source":"!pip install nb-black > /dev/null\n!pip install adjustText > /dev/null","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport warnings\nfrom collections import defaultdict, Counter\n\nwarnings.filterwarnings(\"ignore\")\n\nimport geopandas as gpd\nimport pandas as pd\nimport numpy as np\nimport matplotlib.gridspec as gridspec\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\n\nimport plotly.express as px\nimport plotly.offline as py\nimport plotly.graph_objects as go\n\nfrom adjustText import adjust_text\nfrom plotly.offline import init_notebook_mode, iplot\nfrom wordcloud import WordCloud\n\ninit_notebook_mode(connected=True)\n\nplt.style.use(\"ggplot\")\n\n%load_ext lab_black\n%load_ext autoreload\n%autoreload 2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:22.406441Z","iopub.execute_input":"2022-05-03T03:29:22.406937Z","iopub.status.idle":"2022-05-03T03:29:24.661430Z","shell.execute_reply.started":"2022-05-03T03:29:22.406873Z","shell.execute_reply":"2022-05-03T03:29:24.660410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load data\ntrain = pd.read_csv(\n    \"../input/birdclef-2022-train-metadata-with-audio-metadata/train_ext.csv\"\n)\ntaxonomy = pd.read_csv(\"../input/birdclef-2022/eBird_Taxonomy_v2021.csv\")\nprimary_tax = pd.read_csv(\n    \"../input/the-birds-of-the-hawaiian-islands/primary_checklist_taxonomy.csv\"\n)\nbirdlife = pd.read_csv(\n    \"../input/hbw-and-birdlife-taxonomic-checklist/HBW-BirdLife_List_of_Birds_v6.csv\"\n)\n\n# additional informations\ntrain[\"type\"] = train[\"type\"].apply(eval)\ntrain[\"secondary_labels\"] = train[\"secondary_labels\"].apply(eval)\ntrain_org = train.copy()\n\n# obserbed in Hawaii?\ntrain[\"in_hawaii\"] = (\n    (train[\"longitude\"] >= -161)\n    & (train[\"longitude\"] < -153)\n    & (train[\"latitude\"] >= 18)\n    & (train[\"latitude\"] < 24)\n)\n\n# is_endemic?\nendemic = primary_tax.query(\"'R' in `HAWAIIAN ISLANDS`\")[\"SPECIES_CODE\"].to_numpy()\ntrain[\"is_endemic\"] = train[\"primary_label\"].apply(lambda x: x in endemic)\n\n# extract scored 21 species\nscored_idx = train[\"is_scored\"] == True\nscored = train[scored_idx].reset_index(drop=True)\nscored_org = scored.copy()\n\n# IUCN Red List category\ncategory_map = {\n    \"DD\": 0,\n    \"LC\": 1,\n    \"NT\": 2,\n    \"VU\": 3,\n    \"EN\": 4,\n    \"CR\": 5,\n    \"CR (PE)\": 6,\n    \"EW\": 7,\n    \"EX\": 8,\n}\nbirdlife[\"2021_IUCN_Red_List_category\"] = birdlife[\"2021 IUCN Red List category\"].apply(\n    lambda x: category_map[x]\n)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:24.665667Z","iopub.execute_input":"2022-05-03T03:29:24.666318Z","iopub.status.idle":"2022-05-03T03:29:25.488662Z","shell.execute_reply.started":"2022-05-03T03:29:24.666283Z","shell.execute_reply":"2022-05-03T03:29:25.487754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scored.head(1).T","metadata":{"execution":{"iopub.status.busy":"2022-05-03T03:29:25.489824Z","iopub.execute_input":"2022-05-03T03:29:25.490286Z","iopub.status.idle":"2022-05-03T03:29:25.558872Z","shell.execute_reply.started":"2022-05-03T03:29:25.490249Z","shell.execute_reply":"2022-05-03T03:29:25.557932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"is_scored\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-05-03T03:29:25.561067Z","iopub.execute_input":"2022-05-03T03:29:25.561343Z","iopub.status.idle":"2022-05-03T03:29:25.623764Z","shell.execute_reply.started":"2022-05-03T03:29:25.561311Z","shell.execute_reply":"2022-05-03T03:29:25.622594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Endemic species by Pyle's checklist\n\ndefinition of endemic species is based on Pyle's checklist[1]\n\n* [1] https://www.kaggle.com/datasets/tatamikenn/the-birds-of-the-hawaiian-islands","metadata":{}},{"cell_type":"code","source":"primary_tax.query(\"SPECIES_CODE in @endemic\")[\"PRIMARY_COM_NAME\"].to_numpy()","metadata":{"execution":{"iopub.status.busy":"2022-05-03T03:29:25.626029Z","iopub.execute_input":"2022-05-03T03:29:25.627453Z","iopub.status.idle":"2022-05-03T03:29:25.693050Z","shell.execute_reply.started":"2022-05-03T03:29:25.627398Z","shell.execute_reply":"2022-05-03T03:29:25.692257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Primary concerned 10 species\n- https://www.kaggle.com/code/amandanavine/hawaiian-bird-species/notebook","metadata":{}},{"cell_type":"code","source":"primary_concern = [\n    \"hawgoo\",\n    \"iiwi\",\n    \"crehon\",\n    \"maupar\",\n    \"akiapo\",\n    \"hawcre\",\n    \"hawama\",\n    \"puaioh\",\n    \"hawpet1\",\n    \"barpet\",\n]","metadata":{"execution":{"iopub.status.busy":"2022-05-03T03:29:25.694659Z","iopub.execute_input":"2022-05-03T03:29:25.695242Z","iopub.status.idle":"2022-05-03T03:29:25.756625Z","shell.execute_reply.started":"2022-05-03T03:29:25.695196Z","shell.execute_reply":"2022-05-03T03:29:25.755381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pre-process: remove duplicated secondary labels","metadata":{}},{"cell_type":"code","source":"def check_duplication(df, pkey=\"primary_label\", skey=\"secondary_labels\"):\n    def print_header():\n        print(f\"Duplicated count: {len(duplicated)}\")\n        print(\"\")\n        print(\"filename | primary_label | secondary_labels\")\n        print(\"-\" * 40)\n\n    def print_duplication(filename, primary_label, secondary_labels):\n        print(f\"{filename} | {primary_label} | {secondary_labels}\")\n\n    duplicated = []\n\n    for item in df.itertuples():\n        primary_label = getattr(item, pkey)\n        secondary_labels = getattr(item, skey)\n        if primary_label in set(secondary_labels):\n            duplicated.append((item.filename, primary_label, secondary_labels))\n\n    if len(duplicated) == 0:\n        print(\"no duplication\")\n    else:\n        print_header()\n        for args in duplicated[:5]:\n            print_duplication(*args)\n        print(\"...\")\n        for item in duplicated[-5:]:\n            print_duplication(*args)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-05-03T03:29:25.758195Z","iopub.execute_input":"2022-05-03T03:29:25.758474Z","iopub.status.idle":"2022-05-03T03:29:25.836703Z","shell.execute_reply.started":"2022-05-03T03:29:25.758441Z","shell.execute_reply":"2022-05-03T03:29:25.835536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_duplication(scored_org)","metadata":{"execution":{"iopub.status.busy":"2022-05-03T03:29:25.838419Z","iopub.execute_input":"2022-05-03T03:29:25.838676Z","iopub.status.idle":"2022-05-03T03:29:25.908149Z","shell.execute_reply.started":"2022-05-03T03:29:25.838643Z","shell.execute_reply":"2022-05-03T03:29:25.907099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_duplication(train_org)","metadata":{"execution":{"iopub.status.busy":"2022-05-03T03:29:25.909503Z","iopub.execute_input":"2022-05-03T03:29:25.909727Z","iopub.status.idle":"2022-05-03T03:29:26.024888Z","shell.execute_reply.started":"2022-05-03T03:29:25.909698Z","shell.execute_reply":"2022-05-03T03:29:26.023968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for item in train.itertuples():\n    train.at[item.Index, \"secondary_labels\"] = list(\n        set(item.secondary_labels).difference(set([item.primary_label]))\n    )\nfor item in scored.itertuples():\n    scored.at[item.Index, \"secondary_labels\"] = list(\n        set(item.secondary_labels).difference(set([item.primary_label]))\n    )","metadata":{"execution":{"iopub.status.busy":"2022-05-03T03:29:26.026317Z","iopub.execute_input":"2022-05-03T03:29:26.026584Z","iopub.status.idle":"2022-05-03T03:29:26.379303Z","shell.execute_reply.started":"2022-05-03T03:29:26.026557Z","shell.execute_reply":"2022-05-03T03:29:26.378372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_duplication(scored)","metadata":{"execution":{"iopub.status.busy":"2022-05-03T03:29:26.380504Z","iopub.execute_input":"2022-05-03T03:29:26.380749Z","iopub.status.idle":"2022-05-03T03:29:26.440478Z","shell.execute_reply.started":"2022-05-03T03:29:26.380709Z","shell.execute_reply":"2022-05-03T03:29:26.439336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"check_duplication(train)","metadata":{"execution":{"iopub.status.busy":"2022-05-03T03:29:26.441763Z","iopub.execute_input":"2022-05-03T03:29:26.441995Z","iopub.status.idle":"2022-05-03T03:29:26.560159Z","shell.execute_reply.started":"2022-05-03T03:29:26.441968Z","shell.execute_reply":"2022-05-03T03:29:26.559270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Taxonomy","metadata":{}},{"cell_type":"markdown","source":"Save taxonomy of scored species. I also added URL of [eBird](https://ebird.org) site where the pictures and sound recordings of these classes.","metadata":{}},{"cell_type":"code","source":"train_birds = pd.DataFrame(\n    train.groupby(\"primary_label\").max()[\"is_scored\"]\n).reset_index()\ntrain_birds.rename({\"primary_label\": \"label\"}, axis=1, inplace=True)\n\ntrain_tax = scored_tax = pd.merge(\n    train_birds, taxonomy, left_on=\"label\", right_on=\"SPECIES_CODE\"\n).drop([\"label\", \"TAXON_ORDER\", \"CATEGORY\", \"SPECIES_GROUP\", \"REPORT_AS\"], axis=1)\ntrain_tax[\"URL\"] = scored_tax[\"SPECIES_CODE\"].apply(\n    lambda x: f\"https://ebird.org/species/{x}\"\n)\ntrain_tax[\"is_endemic\"] = train_tax[\"SPECIES_CODE\"].apply(lambda x: x in endemic)\ntrain_tax[\"is_primary_concerned\"] = train_tax[\"SPECIES_CODE\"].apply(\n    lambda x: x in primary_concern\n)\ntrain_tax.to_csv(\"train_taxonomy.csv\", index=False)\nscored_tax = (\n    train_tax.query(\"is_scored == True\")\n    .drop(\"is_scored\", axis=1)\n    .reset_index(drop=True)\n)\n\n# join IUCN Red List category\nscored_tax = pd.merge(scored_tax, taxonomy)\nscored_tax = pd.merge(\n    scored_tax, birdlife, left_on=\"SCI_NAME\", right_on=\"Scientific name\", how=\"left\"\n)\nscored_tax.loc[\n    scored_tax[\"SPECIES_CODE\"] == \"hawcre\", birdlife.columns\n] = birdlife.query(\"`Common name` == 'Hawaii Creeper'\").to_numpy()\n\nscored_tax[\"is_endemic\"] = scored_tax[\"SPECIES_CODE\"].apply(lambda x: x in endemic)\nscored_tax[\"is_primary_concerned\"] = scored_tax[\"SPECIES_CODE\"].apply(\n    lambda x: x in primary_concern\n)\nscored_tax","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:26.563802Z","iopub.execute_input":"2022-05-03T03:29:26.564018Z","iopub.status.idle":"2022-05-03T03:29:26.909899Z","shell.execute_reply.started":"2022-05-03T03:29:26.563991Z","shell.execute_reply":"2022-05-03T03:29:26.908838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Number of endemic species","metadata":{}},{"cell_type":"code","source":"print(\"- {} endemic species in train data.\".format(train_tax[\"is_endemic\"].sum()))\nprint(\"- {} endemic species are scored.\".format(scored_tax[\"is_endemic\"].sum()))","metadata":{"execution":{"iopub.status.busy":"2022-05-03T03:29:26.911440Z","iopub.execute_input":"2022-05-03T03:29:26.911867Z","iopub.status.idle":"2022-05-03T03:29:26.971759Z","shell.execute_reply.started":"2022-05-03T03:29:26.911832Z","shell.execute_reply":"2022-05-03T03:29:26.971093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Note: Order & Family\n\n*Order* is a larger category than *family*[1].\nAll bird orders and famillies are listed in [2].\n\n* [1] https://en.wikipedia.org/wiki/Taxonomic_rank\n* [2] https://en.wikipedia.org/wiki/List_of_birds","metadata":{}},{"cell_type":"code","source":"def taxonomy_count_plot(df1, df2, title, log_scale=False):\n    _, (ax1, ax2) = plt.subplots(1, 2, figsize=(18, 8))\n\n    plt.suptitle(title, fontsize=18)\n    ax1.set_title(\"Order\")\n    ax2.set_title(\"Family\")\n\n    gs = [\n        sns.countplot(\n            y=\"ORDER1\",\n            data=df1,\n            ax=ax1,\n            alpha=0.5,\n            color=\"gray\",\n            order=df1[\"ORDER1\"].value_counts().index,\n            label=\"all species\",\n        ),\n        sns.countplot(\n            y=\"ORDER1\",\n            data=df2,\n            ax=ax1,\n            color=\"orange\",\n            order=df1[\"ORDER1\"].value_counts().index,\n            label=\"scored\",\n        ),\n        sns.countplot(\n            y=\"FAMILY\",\n            data=df1,\n            ax=ax2,\n            color=\"gray\",\n            alpha=0.5,\n            order=df1[\"FAMILY\"].value_counts().index,\n            label=\"all species\",\n        ),\n        sns.countplot(\n            y=\"FAMILY\",\n            data=df2,\n            ax=ax2,\n            color=\"green\",\n            order=df1[\"FAMILY\"].value_counts().index,\n            label=\"scored\",\n        ),\n    ]\n    if log_scale:\n        for g in gs:\n            g.set_xscale(\"log\")\n    ax1.legend()\n    ax2.legend()\n    plt.tight_layout()\n    plt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-05-03T03:29:26.973385Z","iopub.execute_input":"2022-05-03T03:29:26.974132Z","iopub.status.idle":"2022-05-03T03:29:27.069401Z","shell.execute_reply.started":"2022-05-03T03:29:26.974087Z","shell.execute_reply":"2022-05-03T03:29:27.068219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"taxonomy_count_plot(train_tax, scored_tax, \"Distribution of Bird Order and Family\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:27.070923Z","iopub.execute_input":"2022-05-03T03:29:27.071211Z","iopub.status.idle":"2022-05-03T03:29:28.551628Z","shell.execute_reply.started":"2022-05-03T03:29:27.071170Z","shell.execute_reply":"2022-05-03T03:29:28.549610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = pd.merge(\n    train, taxonomy, left_on=\"primary_label\", right_on=\"SPECIES_CODE\", how=\"left\"\n)\ndf2 = pd.merge(\n    scored, taxonomy, left_on=\"primary_label\", right_on=\"SPECIES_CODE\", how=\"left\"\n)\n\ntaxonomy_count_plot(df1, df2, \"Sample Counts of Order and Family\", log_scale=True)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:28.553665Z","iopub.execute_input":"2022-05-03T03:29:28.554229Z","iopub.status.idle":"2022-05-03T03:29:30.770266Z","shell.execute_reply.started":"2022-05-03T03:29:28.554185Z","shell.execute_reply":"2022-05-03T03:29:30.769330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Findings\n\n* About 3/4 of the scored classes are `Passeriformers`[3] order.\n* About 1/2 of the scored classes are `Fringillidae (Finches, Euphonias, and Allies)`[4] family.\n\n\n\n* [3] https://en.wikipedia.org/wiki/Passerine\n* [4] https://en.wikipedia.org/wiki/Finch","metadata":{}},{"cell_type":"markdown","source":"# 2021 IUCN Red List category\n\n- https://www.kaggle.com/datasets/tatamikenn/hbw-and-birdlife-taxonomic-checklist","metadata":{}},{"cell_type":"code","source":"def plot_IUCN_heatap(dfs, axes):\n    for df, ax in zip(dfs, axes):\n        df = df.copy()\n        g = sns.heatmap(\n            df,\n            annot=True,\n            fmt=\"g\",\n            cmap=\"Reds\",\n            ax=ax,\n            vmin=0,\n            vmax=8,\n        )\n        g.set(xlabel=None)\n        g.set(ylabel=None)\n\n\n_, axes = plt.subplots(1, 5, figsize=(25, 6))\n(ax1, ax2, ax3, ax4, ax5) = axes\nplt.suptitle(\"2021 IUCN Red List category in scored species\", fontsize=16)\nax1.set_title(\"all\")\nax2.set_title(\"endemic\")\nax3.set_title(\"non-endemic\")\nax4.set_title(\"primary concerned\")\nax5.set_title(\"non-primary concerned\")\n\nplot_IUCN_heatap(\n    [\n        scored_tax.set_index(\"PRIMARY_COM_NAME\")[[\"2021_IUCN_Red_List_category\"]],\n        scored_tax.query(\"SPECIES_CODE in @endemic\").set_index(\"PRIMARY_COM_NAME\")[\n            [\"2021_IUCN_Red_List_category\"]\n        ],\n        scored_tax.query(\"SPECIES_CODE not in @endemic\").set_index(\"PRIMARY_COM_NAME\")[\n            [\"2021_IUCN_Red_List_category\"]\n        ],\n        scored_tax.query(\"SPECIES_CODE in @primary_concern\").set_index(\n            \"PRIMARY_COM_NAME\"\n        )[[\"2021_IUCN_Red_List_category\"]],\n        scored_tax.query(\"SPECIES_CODE not in @primary_concern\").set_index(\n            \"PRIMARY_COM_NAME\"\n        )[[\"2021_IUCN_Red_List_category\"]],\n    ],\n    axes,\n)\n\nplt.tight_layout()\npd.DataFrame({\"Code\": category_map.keys(), \"Value\": category_map.values()})","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:30.771858Z","iopub.execute_input":"2022-05-03T03:29:30.772746Z","iopub.status.idle":"2022-05-03T03:29:34.092393Z","shell.execute_reply.started":"2022-05-03T03:29:30.772697Z","shell.execute_reply":"2022-05-03T03:29:34.091119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Most of the scored endemic species are considered as \"threatened\" (>= 3) according to IUCN Red List.\n* *Band-rumped Storm Petrel*, *Hawaii Akakihi* are not endangered species, but it is in the primary-concerned list.\n\nNote: according to the host's notebook[1], the common species *Band-rumped Storm-Petrel* is **not** endangered, but its Hawaiian breeding *'Ake'ake* is endangered.\n\n* [1] https://www.kaggle.com/code/amandanavine/hawaiian-bird-species/notebook#'Ake'ake-(Oceanodroma-castro)","metadata":{}},{"cell_type":"markdown","source":"# Geo distribution","metadata":{}},{"cell_type":"code","source":"fig = px.scatter_geo(\n    scored,\n    lat=\"latitude\",\n    lon=\"longitude\",\n    color=\"common_name\",\n    title=\"Geo Distribution\",\n)\nfig.update_geos(lataxis_showgrid=True, lonaxis_showgrid=True)\nfig.update_layout(width=960, height=400, margin={\"r\": 0, \"t\": 30, \"l\": 0, \"b\": 0})\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:34.093891Z","iopub.execute_input":"2022-05-03T03:29:34.094581Z","iopub.status.idle":"2022-05-03T03:29:34.524940Z","shell.execute_reply.started":"2022-05-03T03:29:34.094538Z","shell.execute_reply":"2022-05-03T03:29:34.523938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter_geo(\n    scored.query(\"in_hawaii == True\"),\n    lat=\"latitude\",\n    lon=\"longitude\",\n    color=\"common_name\",\n    title=\"Geo Distribution in Hawaii\",\n)\nfig.update_geos(lataxis_showgrid=True, lonaxis_showgrid=True)\nfig.update_layout(width=960, height=400, margin={\"r\": 0, \"t\": 30, \"l\": 0, \"b\": 0})\nfig.update_layout(\n    geo=dict(\n        projection_scale=45,\n        center=dict(lat=20.5, lon=-157.5),\n    )\n)\nfig.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:34.526251Z","iopub.execute_input":"2022-05-03T03:29:34.526486Z","iopub.status.idle":"2022-05-03T03:29:34.740954Z","shell.execute_reply.started":"2022-05-03T03:29:34.526458Z","shell.execute_reply":"2022-05-03T03:29:34.740243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample Count","metadata":{}},{"cell_type":"code","source":"df = scored.value_counts(\"common_name\")\n\nfig, (ax1, ax2) = plt.subplots(\n    1, 2, figsize=(12, 5), gridspec_kw={\"width_ratios\": [2, 3]}\n)\nsns.countplot(\n    y=\"common_name\",\n    data=scored,\n    order=df.index,\n    ax=ax1,\n    hue=\"is_endemic\",\n    dodge=False,\n)\nax1.set(\n    xlabel=\"Count\",\n    ylabel=\"Common Name\",\n    title=\"Sample Counts\",\n    xscale=\"log\",\n    xlim=(1, 1000),\n)\n\nsns.ecdfplot(df, ax=ax2)\nax2.set(\n    xlim=(0, 100),\n    xlabel=\"Sample count\",\n    title=\"Empirical cumulative distribution of sample count\",\n)\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:33:05.032682Z","iopub.execute_input":"2022-05-03T03:33:05.033039Z","iopub.status.idle":"2022-05-03T03:33:06.004302Z","shell.execute_reply.started":"2022-05-03T03:33:05.033001Z","shell.execute_reply":"2022-05-03T03:33:06.003218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Good number of species are suitable for typical settings of few-shot learning:\n    * **1/3** of species have sample size <= 10\n    * **more than 1/2** of species have sample size <= 20\n\nThus, **10-shot or 20-shot** situation might fit the requirements of the competition.","metadata":{}},{"cell_type":"code","source":"_, axes = plt.subplots(1, 5, figsize=(25, 6))\n(ax1, ax2, ax3, ax4, ax5) = axes\nplt.suptitle(\"Sample count of scored species\", fontsize=16)\nax1.set_title(\"all\")\nax2.set_title(\"endemic\")\nax3.set_title(\"non-endemic\")\nax4.set_title(\"primary concerned\")\nax5.set_title(\"non-primary concerned\")\n\nsns.countplot(y=\"common_name\", data=scored, hue=\"in_hawaii\", ax=ax1)\nsns.countplot(\n    y=\"common_name\",\n    data=scored.query(\"primary_label in @endemic\"),\n    hue=\"in_hawaii\",\n    ax=ax2,\n)\nsns.countplot(\n    y=\"common_name\",\n    data=scored.query(\"primary_label not in @endemic\"),\n    hue=\"in_hawaii\",\n    ax=ax3,\n)\nsns.countplot(\n    y=\"common_name\",\n    data=scored.query(\"primary_label in @primary_concern\"),\n    hue=\"in_hawaii\",\n    ax=ax4,\n)\nsns.countplot(\n    y=\"common_name\",\n    data=scored.query(\"primary_label not in @primary_concern\"),\n    hue=\"in_hawaii\",\n    ax=ax5,\n)\nfor ax in axes:\n    ax.set_ylabel(None)\n\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:35.624072Z","iopub.execute_input":"2022-05-03T03:29:35.624353Z","iopub.status.idle":"2022-05-03T03:29:37.347442Z","shell.execute_reply.started":"2022-05-03T03:29:35.624322Z","shell.execute_reply":"2022-05-03T03:29:37.346322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Most of endemic species have less than 50 samples, some have only a few samples.\n* Almost all of the endemic species are recorded in Hawaii islands. On the other hand, non-endemic species are mostly recorded outside the Hawaii islands.","metadata":{}},{"cell_type":"markdown","source":"# Author","metadata":{}},{"cell_type":"code","source":"_, ax = plt.subplots(figsize=(8, 5))\ndf = scored.value_counts(\"author\") / len(scored) * 100\ndf = df.head(20)\nsns.barplot(x=df, y=df.index, ax=ax)\nax.set(\n    ylabel=\"sample count\",\n    title=\"Number of recording per author (Top 20)\",\n    xlabel=\"Number of recordings (%)\",\n)\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:37.349531Z","iopub.execute_input":"2022-05-03T03:29:37.349895Z","iopub.status.idle":"2022-05-03T03:29:37.765250Z","shell.execute_reply.started":"2022-05-03T03:29:37.349843Z","shell.execute_reply":"2022-05-03T03:29:37.764222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_n = 5\ntop_n_author = 10\ntop_n_species = 8\n\nn_cols = 4\nn_rows = (top_n_species - 1) // n_cols + 1\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(n_cols * 4, n_rows * 3), sharex=True)\naxes = axes.ravel()\ncommon_names = scored.value_counts(\"common_name\").head(top_n_species).index.array\n\nfor ax, common_name in zip(axes, common_names):\n    df_ = scored.query(\"common_name == @common_name\").value_counts(\"author\")\n    df_ = df_ / sum(df_) * 100\n    df = df_.head(top_n_author)\n    sns.barplot(x=df, y=df.index, ax=ax)\n    share = df_.head(top_n).sum()\n    ax.set(\n        title=f\"{common_name} (top {top_n}: {share:.0f}%)\",\n        xlabel=\"recordings (%)\",\n        xlim=(0, 40),\n    )\n\nplt.suptitle(\n    f\"Number of recordings per species (top {top_n_species}), per authors (top {top_n_author})\",\n    fontsize=16,\n)\nplt.tight_layout()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:37.767274Z","iopub.execute_input":"2022-05-03T03:29:37.767716Z","iopub.status.idle":"2022-05-03T03:29:40.341242Z","shell.execute_reply.started":"2022-05-03T03:29:37.767670Z","shell.execute_reply":"2022-05-03T03:29:40.340204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* For top 3 endemic species (Apapane, Iiwi, Omao) are mostly recorded by only a few authors. Most of them also appears in the graph of different species. (Maybe endemic species are only recorded by few number of specialists.)","metadata":{}},{"cell_type":"markdown","source":"# Ratings","metadata":{}},{"cell_type":"code","source":"_, ax = plt.subplots()\nsns.countplot(x=\"rating\", data=scored, ax=ax, color=\"blue\")\nax.set_title(\"Count of Ratings in Scored Classes\")\nax.set_xlabel(\"rating(0-5)\")\nax.set_ylabel(\"count\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:40.342810Z","iopub.execute_input":"2022-05-03T03:29:40.343062Z","iopub.status.idle":"2022-05-03T03:29:40.652293Z","shell.execute_reply.started":"2022-05-03T03:29:40.343031Z","shell.execute_reply":"2022-05-03T03:29:40.651258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\n    \"{:.1f}% of scored recordings are rating >= 3.0.\".format(\n        len(scored[scored[\"rating\"] >= 3.0]) / len(scored) * 100\n    )\n)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:40.653787Z","iopub.execute_input":"2022-05-03T03:29:40.654045Z","iopub.status.idle":"2022-05-03T03:29:40.726674Z","shell.execute_reply.started":"2022-05-03T03:29:40.654011Z","shell.execute_reply":"2022-05-03T03:29:40.725690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, ax = plt.subplots(figsize=(4, 8))\nsns.boxplot(y=\"common_name\", x=\"rating\", data=scored, hue=\"is_endemic\", dodge=False)\nax.set_xlabel(\"rating (0-5)\")\nax.set_title(\"Rating distribution per scored classes\")\nplt.legend(bbox_to_anchor=(1.02, 1), loc=2, borderaxespad=0.0, title=\"is_endemic\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:40.728221Z","iopub.execute_input":"2022-05-03T03:29:40.728540Z","iopub.status.idle":"2022-05-03T03:29:41.367388Z","shell.execute_reply.started":"2022-05-03T03:29:40.728495Z","shell.execute_reply":"2022-05-03T03:29:41.366362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Tags\n\n## Note: Songs & Calls\n\nAccording to Wikipedia[1]\n\n> The distinction between songs and calls is based upon complexity, length, and context. Songs are longer and more complex and are associated with territory and courtship and mating, while calls tend to serve such functions as alarms or keeping members of a flock in contact.\n\n[1] https://en.wikipedia.org/wiki/Bird_vocalization","metadata":{}},{"cell_type":"code","source":"def plot_type_charts(input_df, title):\n    input_df = input_df.copy()\n    types = set(input_df[\"type\"].sum())\n    type_count = defaultdict(int)\n\n    norm_type_count = defaultdict(int)\n\n    print(f\"num types: {len(types)}\")\n    for item in input_df.itertuples():\n        call, song = False, False\n        for t in item.type:\n            t = t.lower()\n            type_count[t] += 1\n            if \"call\" in t:\n                call |= True\n            elif \"song\" in t:\n                song |= True\n            elif \"sing\" in t:\n                song |= True\n            else:\n                pass\n        if call and song:\n            norm_type_count[\"both\"] += 1\n        elif song:\n            norm_type_count[\"song\"] += 1\n        elif call:\n            norm_type_count[\"call\"] += 1\n        else:\n            norm_type_count[\"others\"] += 1\n\n    df = pd.DataFrame({\"type\": type_count.keys(), \"count\": type_count.values()})\n    df.sort_values(\"count\", ascending=False, inplace=True)\n    df_ = pd.DataFrame(\n        {\"type\": norm_type_count.keys(), \"count\": norm_type_count.values()}\n    )\n    df_.sort_values(\"count\", ascending=False, inplace=True)\n\n    _, (ax1, ax2) = plt.subplots(1, 2, figsize=(24, 8))\n    plt.setp(ax1.get_xticklabels(), rotation=90)\n\n    ax1.set_title(\"Broad category of tags\")\n    ax1.pie(\n        df_[\"count\"], labels=df_[\"type\"], autopct=\"%.1f%%\", textprops={\"fontsize\": 16}\n    )\n\n    ax2.set_title(\"Tags in WordCloud\")\n    wordcloud = WordCloud(\n        width=960, height=600, background_color=\"white\"\n    ).generate_from_frequencies(type_count)\n    ax2.imshow(wordcloud)\n    ax2.axis(\"off\")\n\n    plt.suptitle(title, fontsize=20)\n    plt.tight_layout()\n    plt.show()\n\n\nplot_type_charts(scored, \"Tags in scored species\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:41.368843Z","iopub.execute_input":"2022-05-03T03:29:41.369097Z","iopub.status.idle":"2022-05-03T03:29:42.634660Z","shell.execute_reply.started":"2022-05-03T03:29:41.369066Z","shell.execute_reply":"2022-05-03T03:29:42.633535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_type_charts(train, \"Tags in All Train data\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:42.636055Z","iopub.execute_input":"2022-05-03T03:29:42.636325Z","iopub.status.idle":"2022-05-03T03:29:44.716933Z","shell.execute_reply.started":"2022-05-03T03:29:42.636292Z","shell.execute_reply":"2022-05-03T03:29:44.715925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In the charts above, *broad category* is calculated as follows:\n1. when keyword `call` is included in a tag, count it as *call*\n2. when keyword `sing` or `song` is included in a tag, cout it as *song*\n3. in other case, count it as *others*\n\n## Findings\n\n* almost all the broad category are *call* or *song* (or both)\n* Compared to the entire train data, the data of only scored species accounts for a larger percentage of *song*.","metadata":{}},{"cell_type":"markdown","source":"# Sedondary labels","metadata":{}},{"cell_type":"code","source":"fig, ax = plt.subplots()\nsns.countplot(x=\"num_secondary_labels\", data=scored, ax=ax, color=\"blue\")\nax.set_title(\"Countplot of num_secondary_labels\")\nax.set_xlabel(\"number of secondary labels\")\nax.set_ylabel(\"count\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:44.718244Z","iopub.execute_input":"2022-05-03T03:29:44.718491Z","iopub.status.idle":"2022-05-03T03:29:45.001593Z","shell.execute_reply.started":"2022-05-03T03:29:44.718461Z","shell.execute_reply":"2022-05-03T03:29:45.000256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\n    (\"{:.1f}% of scored recordings have no secondary labels.\").format(\n        100 * len(scored.query(\"num_secondary_labels == 0\")) / len(scored)\n    )\n)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:45.004054Z","iopub.execute_input":"2022-05-03T03:29:45.004506Z","iopub.status.idle":"2022-05-03T03:29:45.085429Z","shell.execute_reply.started":"2022-05-03T03:29:45.004447Z","shell.execute_reply":"2022-05-03T03:29:45.083929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, ax = plt.subplots(figsize=(4, 8))\nsns.boxplot(\n    y=\"common_name\",\n    x=\"num_secondary_labels\",\n    data=scored,\n    hue=\"is_endemic\",\n    dodge=False,\n)\nax.set_xlabel(\"number of secondary labels\")\nax.set_title(\"Number of secondary labels per scored classes\")\nplt.legend(bbox_to_anchor=(1.02, 1), loc=2, borderaxespad=0.0, title=\"is_endemic\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:45.087525Z","iopub.execute_input":"2022-05-03T03:29:45.088109Z","iopub.status.idle":"2022-05-03T03:29:45.764592Z","shell.execute_reply.started":"2022-05-03T03:29:45.088051Z","shell.execute_reply":"2022-05-03T03:29:45.763264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The majority of secondary labels are assigned to the endemic species. On the other hand, few are assigned to non-endemic species.","metadata":{}},{"cell_type":"markdown","source":"## Are secondary labels also scored?","metadata":{}},{"cell_type":"code","source":"scored_labels = set(scored[\"primary_label\"].unique())\n\nsl_count = defaultdict(int)\nnon_scored = defaultdict(int)\n\nfor item in scored.itertuples():\n    for label in item.secondary_labels:\n        if label in scored_labels:\n            sl_count[\"scored\"] += 1\n        else:\n            sl_count[\"not_scored\"] += 1\n            non_scored[label] += 1\n\n\ndf = pd.DataFrame({\"label\": sl_count.keys(), \"count\": sl_count.values()})\ndf_ = pd.DataFrame(\n    {\"label\": non_scored.keys(), \"count\": non_scored.values()}\n).sort_values(\"count\", ascending=False)\ndf_ = (\n    pd.merge(\n        df_,\n        train_tax[[\"SPECIES_CODE\", \"PRIMARY_COM_NAME\", \"is_endemic\"]],\n        left_on=\"label\",\n        right_on=\"SPECIES_CODE\",\n    )\n    .drop([\"SPECIES_CODE\", \"label\"], axis=1)\n    .rename({\"PRIMARY_COM_NAME\": \"label\"}, axis=1)\n)\n\n_, (ax1, ax2) = plt.subplots(1, 2, figsize=(20, 6))\n\nax1.set_title(\"Are Secondary Labels Scored?\")\nax1.pie(df[\"count\"], labels=df[\"label\"], autopct=\"%.1f%%\", textprops={\"fontsize\": 16})\n\nax2.set_title(\"counts of non-scored labels\")\nsns.barplot(y=\"label\", x=\"count\", data=df_, hue=\"is_endemic\", dodge=False, ax=ax2)\n\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:45.766499Z","iopub.execute_input":"2022-05-03T03:29:45.767291Z","iopub.status.idle":"2022-05-03T03:29:46.764100Z","shell.execute_reply.started":"2022-05-03T03:29:45.767226Z","shell.execute_reply":"2022-05-03T03:29:46.763012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 2/3 of secondary labels tagged to scored species are also scored species. This is an understandable result considering that most of the secondary labels are added to endemic species.","metadata":{}},{"cell_type":"markdown","source":"## Co-occurrence of second labels","metadata":{}},{"cell_type":"code","source":"data = {\n    \"primary_label\": [],\n    \"secondary_label\": [],\n}\nscored_species = scored_tax[\"SPECIES_CODE\"].unique()\nendemic_species = scored_tax.query(\"is_endemic == True\")[\"SPECIES_CODE\"].unique()\ncode2name = pd.Series(\n    scored_tax.PRIMARY_COM_NAME.values, index=scored_tax.SPECIES_CODE\n).to_dict()\nothers_label = \"*non-scored*\"\n\nfor item in scored.itertuples():\n    p_label = item.primary_label\n    for s_label in item.secondary_labels:\n        data[\"primary_label\"].append(p_label)\n        if s_label in scored_species:\n            data[\"secondary_label\"].append(s_label)\n        else:\n            data[\"secondary_label\"].append(others_label)\n\ncolumns = scored_species.tolist() + [others_label]\ndf = pd.DataFrame(data)\ncross = pd.crosstab(df.primary_label, df.secondary_label)\ncross = cross.reindex(columns, axis=1).fillna(0).astype(int)\ncross_endemic = cross.reset_index().query(\"primary_label in @endemic_species\")\ncross_endemic = cross_endemic.set_index(\"primary_label\")\ncross_endemic = cross_endemic[endemic_species.tolist() + [others_label]]\n\ncross = cross.rename(code2name).rename(code2name, axis=1)\ncross_endemic = cross_endemic.rename(code2name).rename(code2name, axis=1)\n\n# plot\nplt.tight_layout()\nfig = plt.figure(figsize=(16, 13))\ngs0 = gridspec.GridSpec(2, 1, figure=fig)\ngs00 = gridspec.GridSpecFromSubplotSpec(1, 2, subplot_spec=gs0[0])\nax1 = fig.add_subplot(gs00[0])\nax2 = fig.add_subplot(gs00[1])\ngs01 = gridspec.GridSpecFromSubplotSpec(1, 2, subplot_spec=gs0[1])\nax3 = fig.add_subplot(gs01[0])\nax4 = fig.add_subplot(gs01[1])\n\n# ax1\nsns.heatmap(cross.apply(np.log1p), square=True, ax=ax1)\nax1.set(title=\"all scored species\")\n\n# ax2\nsns.heatmap(cross_endemic.apply(np.log1p), square=True, ax=ax2)\nax2.set(title=\"only endemic species\")\n\n# ax3\ndf = cross.sum()\nsns.barplot(y=df.index, x=df, ax=ax3)\nax3.set(title=\"all scored species\")\n\n# ax4\ndf = cross_endemic.sum()\nsns.barplot(y=df.index, x=df, ax=ax4)\nax4.set(title=\"only endemic species\")\n\nplt.suptitle(\n    \"Up: co-occurence matrix (logarithmic scale) / Bottom: distribution of secondary labels\",\n    fontsize=16,\n)\ngs0.tight_layout(fig)\nplt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-05-03T03:29:46.765593Z","iopub.execute_input":"2022-05-03T03:29:46.765953Z","iopub.status.idle":"2022-05-03T03:29:49.002068Z","shell.execute_reply.started":"2022-05-03T03:29:46.765918Z","shell.execute_reply":"2022-05-03T03:29:49.000804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* There is a bias toward certain species that appear on the second labels: Apapane, Hawaii Amakihi, Iiwi are more frequently appeares in background than other scored species.","metadata":{}},{"cell_type":"markdown","source":"## Foreground v.s. background occurrence\n\nBelow is the scatter plot of primary/secondary labels counts. From the graph, we can see that the frequency of the second label is roughly proportional to the frequency of the primary label. If the frequency of the primary labels is proportional to the number of inhabitants of the species, **species that are frequently recorded in the background are simply considered to be those with a high number of inhabitants, or those with larger area of inhabitants**.","metadata":{}},{"cell_type":"code","source":"endemic_count = (\n    pd.DataFrame(scored[\"primary_label\"].value_counts())\n    .reindex(columns)\n    .drop(others_label)\n    .reset_index()\n    .rename(\n        {\"index\": \"primary_label\", \"primary_label\": \"count_of_primary_labels\"}, axis=1\n    )\n    .query(\"primary_label in @endemic_species\")\n)\nendemic_count[\"common_name\"] = endemic_count[\"primary_label\"].apply(\n    lambda x: code2name[x]\n)\n\ndf = pd.merge(\n    endemic_count,\n    cross_endemic.sum().reset_index().rename({0: \"count_of_secondary_labels\"}, axis=1),\n    left_on=\"common_name\",\n    right_on=\"secondary_label\",\n    how=\"left\",\n)\n_, ax = plt.subplots(figsize=(13, 8))\nsns.scatterplot(\n    x=\"count_of_primary_labels\", y=\"count_of_secondary_labels\", data=df, ax=ax\n)\ntexts = [plt.text(X[1], X[4], X[2], ha=\"center\", va=\"center\") for X in df.to_numpy()]\nadjust_text(texts, arrowprops=dict(arrowstyle=\"->\"), color=\"black\")\n\nax.set(\n    title=\"Correlation between #secondary labels and #primary labels\",\n    xlabel=\"count of primary labels\",\n    ylabel=\"count of secondary labels\",\n)\nplt.show()","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-05-03T03:29:49.004011Z","iopub.execute_input":"2022-05-03T03:29:49.004370Z","iopub.status.idle":"2022-05-03T03:29:51.502686Z","shell.execute_reply.started":"2022-05-03T03:29:49.004325Z","shell.execute_reply":"2022-05-03T03:29:51.501737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Audio Statistics\n\n* used extended metadata:\nhttps://www.kaggle.com/tatamikenn/birdclef-2022-train-metadata-with-audio-metadata","metadata":{}},{"cell_type":"code","source":"ax = sns.histplot(x=scored[\"length\"], log_scale=True)\nax.set_xlabel(\"length [sec]\")\nax.set_title(\"Histogram of sequece length in scored classes\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:51.503876Z","iopub.execute_input":"2022-05-03T03:29:51.504638Z","iopub.status.idle":"2022-05-03T03:29:52.484983Z","shell.execute_reply.started":"2022-05-03T03:29:51.504599Z","shell.execute_reply":"2022-05-03T03:29:52.483951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\n    (\"{:.1f}% of recordings are length > 10 sec.\").format(\n        100 * len(scored[scored[\"length\"] > 10]) / len(scored)\n    )\n)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:52.486718Z","iopub.execute_input":"2022-05-03T03:29:52.487328Z","iopub.status.idle":"2022-05-03T03:29:52.566571Z","shell.execute_reply.started":"2022-05-03T03:29:52.487281Z","shell.execute_reply":"2022-05-03T03:29:52.565793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, ax = plt.subplots(figsize=(5, 8))\nax = sns.boxplot(\n    y=\"common_name\", x=\"length\", data=scored, hue=\"is_endemic\", dodge=False\n)\nax.set_xlabel(\"length [sec]\")\nax.set_title(\"Distribution of Sequence Length per Scored Bird Class\")\nax.set_xscale(\"log\")\nplt.legend(bbox_to_anchor=(1.02, 1), loc=2, borderaxespad=0.0, title=\"is_endemic\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:52.568071Z","iopub.execute_input":"2022-05-03T03:29:52.569298Z","iopub.status.idle":"2022-05-03T03:29:53.513825Z","shell.execute_reply.started":"2022-05-03T03:29:52.569239Z","shell.execute_reply":"2022-05-03T03:29:53.512362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, ax = plt.subplots(figsize=(5, 6))\n\n# df = scored.groupby(\"common_name\")[\"length\"].sum().reset_index()\ndf = (\n    scored.groupby(\"common_name\")\n    .agg(length=(\"length\", \"sum\"), is_endemic=(\"is_endemic\", \"max\"))\n    .reset_index()\n)\n\ndf_ = scored.value_counts(\"common_name\")\n\ndf[\"length\"] /= 60\nsns.barplot(\n    y=\"common_name\",\n    x=\"length\",\n    order=df_.index,\n    data=df,\n    ax=ax,\n    hue=\"is_endemic\",\n    dodge=False,\n    alpha=1.0,\n)\nax.set_xlabel(\"length [min]\")\nax.set_title(\"Total Sequence Length of Scored Bird Class\")\nax.set_xscale(\"log\")\nax.set_xlim((1e-1, 1e3))\nplt.show()\nprint(f\"Total length of all scored recordings: {df['length'].sum() / 60:.2f} hours\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:53.516166Z","iopub.execute_input":"2022-05-03T03:29:53.516557Z","iopub.status.idle":"2022-05-03T03:29:54.581022Z","shell.execute_reply.started":"2022-05-03T03:29:53.516504Z","shell.execute_reply":"2022-05-03T03:29:54.579810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Puaiohi has only less than 1 minute recordings.","metadata":{}},{"cell_type":"code","source":"ax = sns.histplot(x=\"channels\", data=scored, bins=2)\nax.set_title(\"Distribution of audio channels for the scored samples\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-05-03T03:29:54.587355Z","iopub.execute_input":"2022-05-03T03:29:54.588396Z","iopub.status.idle":"2022-05-03T03:29:54.919457Z","shell.execute_reply.started":"2022-05-03T03:29:54.588336Z","shell.execute_reply":"2022-05-03T03:29:54.918737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* About 2/3 of recordings are stereo recordings.","metadata":{}},{"cell_type":"markdown","source":"# Visualize audio samples","metadata":{}},{"cell_type":"markdown","source":"## Findings\n\n* Some bird can shift \\~4kHz frequency in a short period of time (\\~100ms), so collectly tuning time/frequency resolution is important.\n* There exists different \"mode\" of call even if it is in the same bird class.\n* Calls or songs of multiple species are in the same recording even if there are no secondary labels. Separating these calls/songs of different bird species is important especially for the bird class of few samples.","metadata":{}},{"cell_type":"code","source":"def get_full_path(path):\n    return f\"../input/birdclef-2022/train_audio/{path}\"\n\n\ndef play_song(filename, bird_class):\n    display(ipd.Audio(get_full_path(filename)))\n\n\ndef load_audio(filename, sr):\n    full_path = get_full_path(filename)\n    assert os.path.isfile(full_path), (full_path, filename)\n    audio, _ = librosa.core.load(full_path, sr=sr, mono=True)\n    return audio\n\n\ndef create_spectrogram(\n    filename,\n    bird_class,\n    audio_params,\n):\n    sr, n_fft, hop_length, n_mels, fmin, fmax = [\n        audio_params[key]\n        for key in [\"sr\", \"n_fft\", \"hop_length\", \"n_mels\", \"fmin\", \"fmax\"]\n    ]\n    audio = load_audio(filename, sr)\n    melspec = librosa.feature.melspectrogram(\n        audio,\n        sr=sr,\n        n_fft=n_fft,\n        hop_length=hop_length,\n        n_mels=n_mels,\n        power=1.0,\n        fmin=fmin,\n        fmax=fmax,\n    )\n    S_db = librosa.amplitude_to_db(melspec, ref=np.max)\n    return S_db\n\n\ndef show_spectrogram(S_db, audio_params, title, fig, ax):\n    hop_length, sr, fmin, fmax = [\n        audio_params[key] for key in [\"hop_length\", \"sr\", \"fmin\", \"fmax\"]\n    ]\n    colormesh = librosa.display.specshow(\n        S_db,\n        hop_length=hop_length,\n        sr=sr,\n        fmin=fmin,\n        fmax=fmax,\n        x_axis=\"time\",\n        y_axis=\"mel\",\n        ax=ax,\n    )\n    ax.set_title(\n        title,\n        fontsize=15,\n    )\n    return colormesh\n\n\ndef view_spectrogram(bird_class, random_state=123, drop_low_freq=False, num_samples=3):\n    audio_params = dict(\n        sr=32_000,\n        n_mels=128,\n        n_fft=800,  # 25 ms\n        hop_length=320,  # 10 ms\n        fmin=0,\n        fmax=16_000,\n    )\n\n    selected = scored.query(\"primary_label == @bird_class\").sort_values(\n        \"length\", ascending=True\n    )\n    common_name = scored_tax.query(\"SPECIES_CODE == @bird_class\")[\n        \"PRIMARY_COM_NAME\"\n    ].array[0]\n    sample_count = min(num_samples, len(selected))\n    print(\"=\" * 40)\n    print(f\"Common Name: {common_name}\")\n    print(f\"URL: https://ebird.org/species/{bird_class}\")\n    print(f\"params: {audio_params}\")\n    print(\"=\" * 40)\n\n    n_fig = len(selected[:sample_count])\n    fig, axes = plt.subplots(n_fig, 1, figsize=(24, 3 * n_fig))\n    for i, item in enumerate(selected[:sample_count].itertuples()):\n        params = {\n            \"rating\": item.rating,\n            \"filename\": item.filename,\n            \"type\": item.type,\n            \"secondary labels\": item.secondary_labels,\n        }\n        print(params)\n        ax = axes[i] if n_fig >= 2 else axes\n\n        title = f\"[Bird: {bird_class}] Mel-scaled spectrogram of audio: {item.filename.split('/')[-1]}\"\n        S_db = create_spectrogram(item.filename, bird_class, audio_params)\n        colormesh = show_spectrogram(S_db, audio_params, title, fig, ax)\n        play_song(item.filename, bird_class)\n\n    plt.tight_layout()\n    fig.colorbar(colormesh, ax=axes, format=\"%+2.0f dB\", location=\"right\")","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"akiapo\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"aniani\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"apapan\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"barpet\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"crehon\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"elepai\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"ercfra\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"hawama\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"hawcre\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"hawgoo\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"hawhaw\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"hawpet1\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"houfin\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"iiwi\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"jabwar\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"maupar\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"omao\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"puaioh\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"skylar\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"warwhe1\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"view_spectrogram(\"yefcan\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}