{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What this notebook does:\n\n* comparing BirdCLEF 2021 train data and BirdCLEF 2022 data","metadata":{}},{"cell_type":"code","source":"!pip install nb-black > /dev/null","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-04-20T14:56:02.901734Z","iopub.execute_input":"2022-04-20T14:56:02.902333Z","iopub.status.idle":"2022-04-20T14:56:14.083402Z","shell.execute_reply.started":"2022-04-20T14:56:02.902200Z","shell.execute_reply":"2022-04-20T14:56:14.082316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import geopandas as gpd\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\n\nfrom matplotlib_venn import venn2\n\nplt.style.use(\"ggplot\")\n%load_ext lab_black","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-20T14:56:14.085795Z","iopub.execute_input":"2022-04-20T14:56:14.086209Z","iopub.status.idle":"2022-04-20T14:56:15.836443Z","shell.execute_reply.started":"2022-04-20T14:56:14.086165Z","shell.execute_reply":"2022-04-20T14:56:15.835565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load train data\ntrain_2021 = pd.read_csv(\"../input/birdclef-2021/train_metadata.csv\").drop(\n    \"date\", axis=1\n)\ntrain_2022 = pd.read_csv(\"../input/birdclef-2022/train_metadata.csv\")\nscored = pd.read_json(\"../input/birdclef-2022/scored_birds.json\")\n\n# normalize columns of 2021\ntrain_2021 = train_2021.reindex(train_2022.columns, axis=1)  # normalize column order\ntrain_2021[\"filename\"] = train_2021[\"primary_label\"] + \"/\" + train_2021[\"filename\"]\nassert (train_2021.columns == train_2022.columns).all()\n\n# add year columns\ntrain_2021[\"year\"] = 2021\ntrain_2022[\"year\"] = 2022\n\n# append audio metadata\naudio_2021 = pd.read_csv(\n    \"../input/birdclef-2022-train-metadata-with-audio-metadata/audio_metadata_2021.csv\"\n)\naudio_2022 = pd.read_csv(\n    \"../input/birdclef-2022-train-metadata-with-audio-metadata/audio_metadata_2022.csv\"\n)\ntrain_2021 = pd.concat([train_2021, audio_2021], axis=1)\ntrain_2022 = pd.concat([train_2022, audio_2022], axis=1)\n\n# concat 2021 and 2022\ntrain = pd.concat([train_2021, train_2022])\nassert len(train) == len(train_2021) + len(train_2022)\n\n# add auxiliary columns\nscored[\"is_scored\"] = True\nscored.rename({0: \"label\"}, axis=1, inplace=True)\ntrain = (\n    pd.merge(train, scored, left_on=\"primary_label\", right_on=\"label\", how=\"left\")\n    .fillna(False)\n    .drop(\"label\", axis=1)\n)\n# num_secondary_labels\ntrain[\"num_secondary_labels\"] = train[\"secondary_labels\"].apply(lambda x: len(eval(x)))\n\n# re-split into 2021 and 2022\ntrain_2021, train_2022 = train.head(len(train_2021)), train.tail(len(train_2022))\n\n# distinct dataset\ntrain_distinct = train.drop_duplicates(subset=[\"filename\"])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-20T14:56:15.838270Z","iopub.execute_input":"2022-04-20T14:56:15.838566Z","iopub.status.idle":"2022-04-20T14:56:17.140339Z","shell.execute_reply.started":"2022-04-20T14:56:15.838526Z","shell.execute_reply":"2022-04-20T14:56:17.139527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Duplication Check","metadata":{}},{"cell_type":"code","source":"\"scored classes in 2021 data: {}\".format(\n    train_2021[train_2021[\"is_scored\"] == True][\"primary_label\"].unique()\n)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-20T14:56:17.142436Z","iopub.execute_input":"2022-04-20T14:56:17.142740Z","iopub.status.idle":"2022-04-20T14:56:17.157482Z","shell.execute_reply.started":"2022-04-20T14:56:17.142701Z","shell.execute_reply":"2022-04-20T14:56:17.156388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, (ax1, ax2) = plt.subplots(1, 2, figsize=(10, 5))\nplt.suptitle(\"Duplication between 2021 vs 2022\", fontsize=16)\nplt.tight_layout()\n\nl21, l22 = set(train_2021[\"primary_label\"].unique()), set(\n    train_2022[\"primary_label\"].unique()\n)\nvenn2(subsets=(l21, l22), set_labels=(\"train_2021\", \"train_2022\"), ax=ax1)\nax1.set_title(\"primary_labels\")\n\nf21, f22 = set(train_2021[\"filename\"].unique()), set(train_2022[\"filename\"].unique())\nvenn2(subsets=(f21, f22), set_labels=(\"train_2021\", \"train_2022\"), ax=ax2)\nax2.set_title(\"filename\")\n\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-20T14:56:17.158855Z","iopub.execute_input":"2022-04-20T14:56:17.159196Z","iopub.status.idle":"2022-04-20T14:56:17.511898Z","shell.execute_reply.started":"2022-04-20T14:56:17.159145Z","shell.execute_reply":"2022-04-20T14:56:17.511009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* 40 bird species and 5901 files are duplicated in 2021 and 2022 data.","metadata":{}},{"cell_type":"markdown","source":"# Geo Distribution","metadata":{}},{"cell_type":"code","source":"_, ax = plt.subplots(figsize=(13, 8))\n\ncountries = gpd.read_file(gpd.datasets.get_path(\"naturalearth_lowres\"))\ncountries.plot(color=\"lightgrey\", ax=ax)\nsns.scatterplot(\n    x=\"longitude\",\n    y=\"latitude\",\n    data=train,\n    hue=\"year\",\n    palette=\"Set1\",\n    alpha=0.5,\n    marker=\"+\",\n    ax=ax,\n)\n\nax.set_title(\"Geo Distribution\")\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-20T14:56:17.513292Z","iopub.execute_input":"2022-04-20T14:56:17.513595Z","iopub.status.idle":"2022-04-20T14:56:19.221022Z","shell.execute_reply.started":"2022-04-20T14:56:17.513554Z","shell.execute_reply":"2022-04-20T14:56:19.220030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Taxonomy","metadata":{}},{"cell_type":"code","source":"def create_tax_df(taxonomy, labels):\n    labels = list(labels)\n    birds = pd.DataFrame({\"label\": labels})\n\n    tax = pd.merge(birds, taxonomy, left_on=\"label\", right_on=\"SPECIES_CODE\").drop(\n        [\"label\", \"TAXON_ORDER\", \"CATEGORY\", \"SPECIES_GROUP\", \"REPORT_AS\"], axis=1\n    )\n    tax[\"URL\"] = tax[\"SPECIES_CODE\"].apply(lambda x: f\"https://ebird.org/species/{x}\")\n    return tax\n\n\ntaxonomy = pd.read_csv(\"../input/birdclef-2022/eBird_Taxonomy_v2021.csv\")\ntax_2021 = create_tax_df(taxonomy, l21 - l22)\ntax_2021[\"year\"] = 2021\ntax_2022 = create_tax_df(taxonomy, l22)\ntax_2022[\"year\"] = 2022\ntax_merged = pd.concat([tax_2021, tax_2022])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-20T14:56:19.222239Z","iopub.execute_input":"2022-04-20T14:56:19.222459Z","iopub.status.idle":"2022-04-20T14:56:19.334588Z","shell.execute_reply.started":"2022-04-20T14:56:19.222433Z","shell.execute_reply":"2022-04-20T14:56:19.333410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def taxonomy_count_plot(\n    df1, df2, title, label1=\"2021 + 2022\", label2=\"2022\", log_scale=False\n):\n    _, (ax1, ax2) = plt.subplots(1, 2, figsize=(18, 14))\n\n    plt.suptitle(title, fontsize=18)\n    ax1.set_title(\"Order\")\n    ax2.set_title(\"Family\")\n\n    gs = [\n        sns.countplot(\n            y=\"ORDER1\",\n            data=df1,\n            ax=ax1,\n            alpha=0.5,\n            color=\"gray\",\n            order=df1[\"ORDER1\"].value_counts().index,\n            label=label1,\n        ),\n        sns.countplot(\n            y=\"ORDER1\",\n            data=df2,\n            ax=ax1,\n            color=\"orange\",\n            order=df1[\"ORDER1\"].value_counts().index,\n            label=label2,\n        ),\n        sns.countplot(\n            y=\"FAMILY\",\n            data=df1,\n            ax=ax2,\n            color=\"gray\",\n            alpha=0.5,\n            order=df1[\"FAMILY\"].value_counts().index,\n            label=label1,\n        ),\n        sns.countplot(\n            y=\"FAMILY\",\n            data=df2,\n            ax=ax2,\n            color=\"green\",\n            order=df1[\"FAMILY\"].value_counts().index,\n            label=label2,\n        ),\n    ]\n    if log_scale:\n        for g in gs:\n            g.set_xscale(\"log\")\n    ax1.legend()\n    ax2.legend()\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-04-20T14:59:52.878112Z","iopub.execute_input":"2022-04-20T14:59:52.878410Z","iopub.status.idle":"2022-04-20T14:59:52.912699Z","shell.execute_reply.started":"2022-04-20T14:59:52.878375Z","shell.execute_reply":"2022-04-20T14:59:52.911984Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"taxonomy_count_plot(tax_merged, tax_2022, \"Distribution of Bird Order and Family\")","metadata":{"execution":{"iopub.status.busy":"2022-04-20T14:59:53.037338Z","iopub.execute_input":"2022-04-20T14:59:53.037942Z","iopub.status.idle":"2022-04-20T14:59:55.510199Z","shell.execute_reply.started":"2022-04-20T14:59:53.037894Z","shell.execute_reply":"2022-04-20T14:59:55.509499Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1 = pd.merge(\n    train_distinct,\n    taxonomy,\n    left_on=\"primary_label\",\n    right_on=\"SPECIES_CODE\",\n    how=\"left\",\n)\ndf2 = pd.merge(\n    train_2022, taxonomy, left_on=\"primary_label\", right_on=\"SPECIES_CODE\", how=\"left\"\n)\n\ntaxonomy_count_plot(df1, df2, \"Sample Counts of Order and Family\", log_scale=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-20T14:59:55.511342Z","iopub.execute_input":"2022-04-20T14:59:55.512050Z","iopub.status.idle":"2022-04-20T14:59:58.926793Z","shell.execute_reply.started":"2022-04-20T14:59:55.512020Z","shell.execute_reply":"2022-04-20T14:59:58.925523Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample Counts","metadata":{}},{"cell_type":"code","source":"Sltfig, (ax1) = plt.subplots(1, 1, figsize=(8, 3))\ndf1 = train_2021[\"primary_label\"].value_counts()\ndf2 = train_2022[\"primary_label\"].value_counts()\nax1.bar(\n    x=list(range(len(df1))),\n    height=df1.values,\n    color=\"blue\",\n    width=1,\n    alpha=0.8,\n    label=\"2021\",\n)\nax1.bar(\n    x=list(range(len(df2))),\n    height=df2.values,\n    color=\"red\",\n    width=1,\n    alpha=0.8,\n    label=\"2022\",\n)\n\nax1.set(\n    title=f\"distribution of sample counts per species\",\n    xticks=[],\n    xlabel=\"Species\",\n    ylabel=\"Count\",\n)\nax1.legend()\nplt.tight_layout()\nplt.show()\nprint(\"total sample counts:\")\nprint(f\"  - 2021: {len(train_2021)}\")\nprint(f\"  - 2022: {len(train_2022)}\")\nprint(\"unique species:\")\nprint(f\"  - 2021: {train_2021['primary_label'].nunique()}\")\nprint(f\"  - 2022: {train_2022['primary_label'].nunique()}\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-20T15:01:45.992312Z","iopub.execute_input":"2022-04-20T15:01:45.992612Z","iopub.status.idle":"2022-04-20T15:01:47.335110Z","shell.execute_reply.started":"2022-04-20T15:01:45.992586Z","shell.execute_reply":"2022-04-20T15:01:47.334146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, (ax1, ax2) = plt.subplots(1, 2, figsize=(13, 5), sharex=True)\ndf1 = (\n    train_distinct.query(\"year == 2021\")\n    .groupby(\"primary_label\")\n    .agg(count=(\"year\", \"count\"))\n)\ndf2 = (\n    train_distinct.query(\"year == 2022\")\n    .groupby(\"primary_label\")\n    .agg(count=(\"year\", \"count\"))\n)\n\nsns.histplot(x=\"count\", data=df1, log_scale=True, ax=ax1, color=\"blue\")\nsns.histplot(x=\"count\", data=df2, log_scale=True, ax=ax2, color=\"red\")\nax1.set(xlabel=\"Sample count\", title=\"2021\", xlim=(1e-1, 1e3))\nax2.set(xlabel=\"Sample count\", title=\"2022\", xlim=(1e-1, 1e3))\nplt.suptitle(\"Distribution of sample counts per species\", fontsize=16)\nplt.tight_layout()\nplt.show()\nprint(\"* mean sample counts per species:\")\nprint(f\"  - 2021: {df1['count'].mean():.1f}\")\nprint(f\"  - 2022: {df2['count'].mean():.1f}\")\n\nprint(\"* number of species with sample counts >= 20:\")\nprint(f\"  - 2021: {len(df1[df1['count'] >= 20])}/{len(df1)}\")\nprint(f\"  - 2022: {len(df2[df2['count'] >= 20])}/{len(df2)}\")","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-04-20T15:01:47.336533Z","iopub.execute_input":"2022-04-20T15:01:47.336765Z","iopub.status.idle":"2022-04-20T15:01:48.204363Z","shell.execute_reply.started":"2022-04-20T15:01:47.336736Z","shell.execute_reply":"2022-04-20T15:01:48.203511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ratings","metadata":{}},{"cell_type":"code","source":"_, (ax1, ax2) = plt.subplots(1, 2, figsize=(13, 5))\ndf1 = train_distinct.query(\"year == 2021\")\ndf2 = train_distinct.query(\"year == 2022\")\n\nsns.countplot(\n    x=\"rating\",\n    data=df1,\n    color=\"blue\",\n    label=\"2021\",\n    ax=ax1,\n)\n\nsns.countplot(\n    x=\"rating\",\n    data=df2,\n    color=\"red\",\n    label=\"2022\",\n    ax=ax2,\n)\nax1.set(xlabel=\"Rating\", title=\"2021\")\nax2.set(xlabel=\"Rating\", title=\"2022\")\nplt.suptitle(\"Distribution of rating\", fontsize=16)\nplt.tight_layout()\nplt.show()\nprint(\"mean rating:\")\nprint(\n    f\"  - 2021: {df1['rating'].mean():.1f} sec. (std: {df1['rating'].std():.1f} sec.)\"\n)\nprint(\n    f\"  - 2022: {df2['rating'].mean():.1f} sec. (std: {df2['rating'].std():.1f} sec.)\"\n)","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-04-20T15:01:48.205684Z","iopub.execute_input":"2022-04-20T15:01:48.205988Z","iopub.status.idle":"2022-04-20T15:01:48.785520Z","shell.execute_reply.started":"2022-04-20T15:01:48.205947Z","shell.execute_reply":"2022-04-20T15:01:48.784488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Secondary Labels","metadata":{}},{"cell_type":"code","source":"_, (ax1, ax2) = plt.subplots(1, 2, figsize=(13, 5))\ndf1 = train_distinct.query(\"year == 2021\")\ndf2 = train_distinct.query(\"year == 2022\")\n\nsns.countplot(\n    x=\"num_secondary_labels\",\n    data=df1,\n    color=\"blue\",\n    label=\"2021\",\n    ax=ax1,\n)\n\nsns.countplot(\n    x=\"num_secondary_labels\",\n    data=df2,\n    color=\"red\",\n    label=\"2022\",\n    ax=ax2,\n)\nax1.set(xlabel=\"#secondary labels\", title=\"2021\")\nax2.set(xlabel=\"#secondary labels\", title=\"2022\")\nplt.suptitle(\"Distribution of #secondary labels\", fontsize=16)\nplt.tight_layout()\nplt.show()\nprint(\"mean #secondary labels:\")\nprint(\n    f\"  - 2021: {df1['num_secondary_labels'].mean():.1f} (std: {df1['num_secondary_labels'].std():.1f})\"\n)\nprint(\n    f\"  - 2022: {df2['num_secondary_labels'].mean():.1f} (std: {df2['num_secondary_labels'].std():.1f})\"\n)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-20T15:01:48.787178Z","iopub.execute_input":"2022-04-20T15:01:48.787695Z","iopub.status.idle":"2022-04-20T15:01:49.181639Z","shell.execute_reply.started":"2022-04-20T15:01:48.787659Z","shell.execute_reply":"2022-04-20T15:01:49.180604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Audio Length","metadata":{}},{"cell_type":"markdown","source":"## Audio length per sample","metadata":{}},{"cell_type":"code","source":"_, (ax1, ax2) = plt.subplots(1, 2, figsize=(13, 5), sharex=True)\ndf1 = train_distinct.query(\"year == 2021\")\ndf2 = train_distinct.query(\"year == 2022\")\n\nsns.histplot(x=\"length\", data=df1, log_scale=True, ax=ax1, color=\"blue\")\n\nsns.histplot(x=\"length\", data=df2, log_scale=True, ax=ax2, color=\"red\")\nax1.set(xlabel=\"length [sec]\", title=\"2021\")\nax2.set(xlabel=\"length [sec]\", title=\"2022\")\nplt.suptitle(\"Distribution of audio length per sample\", fontsize=16)\nplt.tight_layout()\nplt.show()\nprint(\"mean length of one clip:\")\nprint(f\"  - 2021: {df1['length'].mean():.1f} sec.\")\nprint(f\"  - 2022: {df2['length'].mean():.1f} sec.\")\nprint(\"total length:\")\nprint(f\"  - 2021: {df1['length'].sum() / 3600:.0f} hours\")\nprint(f\"  - 2022: {df2['length'].sum() / 3600:.0f} hours\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-20T15:01:49.184059Z","iopub.execute_input":"2022-04-20T15:01:49.184386Z","iopub.status.idle":"2022-04-20T15:01:50.345547Z","shell.execute_reply.started":"2022-04-20T15:01:49.184342Z","shell.execute_reply":"2022-04-20T15:01:50.344633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* The 2021 data is clipped with a **lower bound of 6 seconds**.\n* The total length of audio samples in 2021 data are about **1/10** of that in 2021.","metadata":{}},{"cell_type":"markdown","source":"## Total audio length per species","metadata":{}},{"cell_type":"code","source":"_, (ax1, ax2) = plt.subplots(1, 2, figsize=(13, 5), sharex=True)\ndf1 = (\n    train_distinct.query(\"year == 2021\").groupby(\"primary_label\")[[\"length\"]].sum() / 60\n)\ndf2 = (\n    train_distinct.query(\"year == 2022\").groupby(\"primary_label\")[[\"length\"]].sum() / 60\n)\n\nsns.histplot(x=\"length\", data=df1, log_scale=True, ax=ax1, color=\"blue\")\n\nsns.histplot(x=\"length\", data=df2, log_scale=True, ax=ax2, color=\"red\")\nax1.set(xlabel=\"length [min]\", title=\"2021\", xlim=(1e-1, 1e3))\nax2.set(xlabel=\"length [min]\", title=\"2022\", xlim=(1e-1, 1e3))\nplt.suptitle(\"Distribution of total audio length per species\", fontsize=16)\nplt.tight_layout()\nplt.show()\nprint(\"mean total length per species:\")\nprint(f\"  - 2021: {df1['length'].mean():.1f} min.\")\nprint(f\"  - 2022: {df2['length'].mean():.1f} min.\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-04-20T15:01:50.383608Z","iopub.execute_input":"2022-04-20T15:01:50.384231Z","iopub.status.idle":"2022-04-20T15:01:51.185904Z","shell.execute_reply.started":"2022-04-20T15:01:50.384190Z","shell.execute_reply":"2022-04-20T15:01:51.185004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* mean total audio length per species in 2022 are about **28%** of that in 2021.","metadata":{}},{"cell_type":"markdown","source":"# Audio Channels","metadata":{}},{"cell_type":"code","source":"_, (ax1, ax2) = plt.subplots(1, 2, figsize=(13, 5), sharex=True)\ndf1 = train_distinct.query(\"year == 2021\")\ndf2 = train_distinct.query(\"year == 2022\")\n\nsns.countplot(\n    x=\"channels\",\n    data=df1,\n    color=\"blue\",\n    label=\"2021\",\n    ax=ax1,\n)\n\nsns.countplot(\n    x=\"channels\",\n    data=df2,\n    color=\"red\",\n    label=\"2022\",\n    ax=ax2,\n)\nax1.set(title=\"2021\")\nax2.set(title=\"2022\")\nplt.suptitle(\"Distribution of channels\", fontsize=16)\nplt.tight_layout()\nplt.show()\nprint(\"mean channels:\")\nprint(f\"  - 2021: {df1['channels'].mean():.1f} (std: {df1['channels'].std():.1f})\")\nprint(f\"  - 2022: {df2['channels'].mean():.1f} (std: {df2['channels'].std():.1f})\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-03-29T14:39:59.660891Z","iopub.execute_input":"2022-03-29T14:39:59.661125Z","iopub.status.idle":"2022-03-29T14:40:00.007559Z","shell.execute_reply.started":"2022-03-29T14:39:59.661097Z","shell.execute_reply":"2022-03-29T14:40:00.006641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sample Rate","metadata":{}},{"cell_type":"code","source":"df1 = train_distinct.query(\"year == 2021\")\ndf2 = train_distinct.query(\"year == 2022\")\n\ndf1[\"sample_rate\"].value_counts(), df2[\"sample_rate\"].value_counts(),","metadata":{"execution":{"iopub.status.busy":"2022-03-29T14:40:00.009624Z","iopub.execute_input":"2022-03-29T14:40:00.010125Z","iopub.status.idle":"2022-03-29T14:40:00.046529Z","shell.execute_reply.started":"2022-03-29T14:40:00.010081Z","shell.execute_reply":"2022-03-29T14:40:00.045845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* both the 2021's and 2022's data are sampled with 32kHz.","metadata":{}}]}