{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30407,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Here is my initial exploration of the dataset. I'll continue to update it as the competition goes on.","metadata":{}},{"cell_type":"markdown","source":"# Takeaways\n\n- Training data contains 24,459 recordings in total, 284.8 hours, 122 GB uncompressed (32 khz with 4 bytes per frame)\n- There are 182 unique primary labels, 140 unique secondary labels\n- Only 1892 or 7% of the recordings have secondary labels","metadata":{}},{"cell_type":"markdown","source":"# Training metadata","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\ndf","metadata":{"execution":{"iopub.status.busy":"2024-04-03T23:52:01.071682Z","iopub.execute_input":"2024-04-03T23:52:01.072396Z","iopub.status.idle":"2024-04-03T23:52:01.236026Z","shell.execute_reply.started":"2024-04-03T23:52:01.072334Z","shell.execute_reply":"2024-04-03T23:52:01.234859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchaudio\nimport plotly.express as px\nfrom IPython.display import Audio\n\ntrain_path = '/kaggle/input/birdclef-2024/train_audio/'\ndata, rate = torchaudio.load(train_path + df.filename[0])\ndisplay(Audio(data[0, :rate*5], rate=rate))\npx.line(y=data[0, :rate*5], title=df.common_name[0])","metadata":{"execution":{"iopub.status.busy":"2024-04-03T23:55:09.871659Z","iopub.execute_input":"2024-04-03T23:55:09.872285Z","iopub.status.idle":"2024-04-03T23:55:11.531607Z","shell.execute_reply.started":"2024-04-03T23:55:09.872230Z","shell.execute_reply":"2024-04-03T23:55:11.529904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Primary labels","metadata":{}},{"cell_type":"code","source":"import plotly.express as px\n\nprimary_label_counts = df.primary_label.value_counts()\n\npx.bar(\n    x=primary_label_counts.keys(), \n    y=primary_label_counts.values,\n    title=\"Distribution of primary labels\",\n    labels={\"x\": \"bird\", \"y\": \"# of recordings\"},\n).show()\n\nprint('# of primary labels:', len(primary_label_counts))","metadata":{"execution":{"iopub.status.busy":"2024-04-03T23:55:25.682061Z","iopub.execute_input":"2024-04-03T23:55:25.682562Z","iopub.status.idle":"2024-04-03T23:55:25.776958Z","shell.execute_reply.started":"2024-04-03T23:55:25.682519Z","shell.execute_reply":"2024-04-03T23:55:25.775299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from shapely.geometry import Point\nimport geopandas as gpd\n\ngeometry = [Point(xy) for xy in zip(df['longitude'], df['latitude'])]\ngdf = gpd.GeoDataFrame(df, geometry=geometry)\nworld = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres'))\nax = world.plot(figsize=(10, 6))\nax.set_axis_off()\nax.set_title('Distribution of recordings')\ngdf.plot(ax=ax, marker='o', color='pink', markersize=1);","metadata":{"execution":{"iopub.status.busy":"2024-04-03T23:55:26.599568Z","iopub.execute_input":"2024-04-03T23:55:26.601162Z","iopub.status.idle":"2024-04-03T23:55:33.563679Z","shell.execute_reply.started":"2024-04-03T23:55:26.601086Z","shell.execute_reply":"2024-04-03T23:55:33.561935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Secondary labels","metadata":{}},{"cell_type":"code","source":"secondary_labels = df[df.secondary_labels != '[]'].reset_index()\nprint('# of recordings with secondary labels:', len(secondary_labels), 'which is', 100 * len(secondary_labels) / len(df), '%')\nsecondary_labels","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels = [e[2:-2].split(\"', '\") for e in secondary_labels.secondary_labels]\nlabels = [e for li in labels for e in li]\nsecondary_label_counts = pd.DataFrame({'secondary_label': labels}).secondary_label.value_counts()\n\npx.bar(\n    x=secondary_label_counts.keys(), \n    y=secondary_label_counts.values,\n    title=\"Distribution of secondary labels\",\n    labels={\"x\": \"bird\", \"y\": \"# of recordings\"},\n).show()\n\nprint('# of secondary labels:', len(secondary_label_counts))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"primarey_secondary_counts_df = pd.concat([\n    pd.DataFrame({\n        \"label\": primary_label_counts.keys(), \n        \"num_recordings\": primary_label_counts.values,\n        \"type\": \"primary\"\n    }),\n    pd.DataFrame({\n        \"label\": secondary_label_counts.keys(), \n        \"num_recordings\": secondary_label_counts.values,\n        \"type\": \"secondary\"\n    })\n])\n\npx.bar(\n    data_frame=primarey_secondary_counts_df,\n    x=\"label\", \n    y=\"num_recordings\",\n    color=\"type\",\n    title=\"Distribution of primary and secondary labels\",\n    barmode=\"group\",\n    labels={\"num_recordings\": \"# of recordings\"}\n).show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_birds = pd.DataFrame({\n    \"label\": primary_label_counts.keys(), \n    \"num_primary_recordings\": primary_label_counts.values,\n}).set_index('label').join(\n    pd.DataFrame({\n        \"label\": secondary_label_counts.keys(), \n        \"num_secondary_recordings\": secondary_label_counts.values,\n    }).set_index('label'),\n)\n\nprint('birds that are more frequent as secondary labels:')\nall_birds[all_birds.num_primary_recordings < all_birds.num_secondary_recordings]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Types","metadata":{}},{"cell_type":"code","source":"types = [e[2:-2].split(\"', '\") for e in df[(df.type != '[]') & (df.type != \"['']\")].type]\ntypes = [e for li in types for e in li]\ntype_counts = pd.DataFrame({'type': types}).type.value_counts()\n\nn = 30\ntop_types = type_counts[:n]\npx.bar(\n    x=top_types.keys(), \n    y=top_types.values,\n    title=f\"Distribution of {n} most common recording types\",\n    labels={\"x\": \"type\", \"y\": \"# of recordings\"},\n).show()\n\nprint('# of distinct types:', len(type_counts))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"has_call = df.type.str.contains('call', case=False)\nhas_song = df.type.str.contains('song', case=False)\n\ndef get_call_song(c, s):\n    if c and s:\n        return 'both'\n    elif c:\n        return 'call'\n    elif s:\n        return 'song'\n    return 'neither'\n\ndf['call_song'] = [get_call_song(has_call[i], has_song[i]) for i in range(len(has_call))]\n\ncall_song_counts = df.call_song.value_counts()\npx.bar(\n    x=call_song_counts.keys(), \n    y=call_song_counts.values / len(df) * 100,\n    title=f\"% of recordings with song and/or call types\",\n    labels={\"x\": \"type\", \"y\": \"% of recordings\"},\n).show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Audio length","metadata":{}},{"cell_type":"code","source":"import torchaudio\nfrom tqdm import tqdm\nfrom joblib import Parallel, delayed\nimport os\n\nmetadatas = Parallel(n_jobs=os.cpu_count())(\n    delayed(lambda filename: torchaudio.info(train_path + filename))(filename) \n    for filename in tqdm(df.filename)\n)\n    \ndf['num_frames'] = [m.num_frames for m in metadatas]\n\n(\n    set([m.sample_rate for m in metadatas]),\n    set([m.encoding for m in metadatas]),\n    set([m.num_channels for m in metadatas]),\n    set([m.bits_per_sample for m in metadatas]),\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_rate = metadatas[0].sample_rate\nnum_samples = df['num_frames'].sum()\nnum_hours = num_samples / sample_rate / 60 / 60\nmax_min = df['num_frames'].max() / sample_rate / 60\nprint('totale # of samples:', num_samples)\nprint('total hours:', num_hours)\nminutes = df['num_frames'] / sample_rate / 60\nminutes.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.histogram(pd.DataFrame({\"minutes\": minutes}), x=\"minutes\", title=\"Distribution of recording lengths (minutes)\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"primary_label_frames = df.groupby('primary_label').num_frames.sum()\n\nprimarey_secondary_counts_df = pd.concat([\n    pd.DataFrame({\n        \"label\": primary_label_counts.keys(), \n        \"percent\": 100 * primary_label_counts.values / primary_label_counts.values.sum(),\n        \"aggregation\": \"num_recordings\"\n    }),\n    pd.DataFrame({\n        \"label\": primary_label_frames.keys(), \n        \"percent\": 100 * primary_label_frames.values / primary_label_frames.values.sum(),\n        \"aggregation\": \"num_samples\"\n    })\n])\n\npx.bar(\n    data_frame=primarey_secondary_counts_df,\n    x=\"label\", \n    y=\"percent\",\n    color=\"aggregation\",\n    title=\"Distribution of primary labels, # of recordings and sum of samples\",\n    barmode=\"group\",\n).show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data, rate = torchaudio.load(train_path + df.filename.iloc[0])\nbytes_per_sample = data.element_size()\ntotal_gigs = df.num_frames.sum() * bytes_per_sample / 2**30\ntotal_gigs","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"info = torchaudio.info(train_path + df.filename.iloc[0])\ninfo.bits_per_sample, info.encoding, info.num_channels, info.num_frames, info.sample_rate","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# How filtering effects dataset size","metadata":{}},{"cell_type":"code","source":"import numpy as np\n\npx.line(\n    title='Accumulated size of dataset (sorting by filesize)',\n    y=np.cumsum(sorted(df.num_frames * bytes_per_sample)) / 2**30,\n    labels={\"x\": \"# of files\", \"y\": \"Dataset size (GB)\"},\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"line = px.line(\n    np.cumsum(bytes_per_sample * df.groupby('rating').num_frames.sum()[::-1]) / 2 ** 30,\n    title='Accumulated size of dataset (sorting by rating, reversed)',\n    labels={\"value\": \"Dataset size (GB)\"},\n)\nline.layout.update(showlegend=False)\nline.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Saving metadata","metadata":{}},{"cell_type":"code","source":"df.to_csv('./train_metadata_with_num_frames.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}