{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport json\nimport torch\nimport matplotlib.pyplot as plt\nimport torchaudio\n\nBASE_DIR = '../input/birdclef-2022/'","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-06T20:32:18.008116Z","iopub.execute_input":"2022-03-06T20:32:18.008457Z","iopub.status.idle":"2022-03-06T20:32:19.695187Z","shell.execute_reply.started":"2022-03-06T20:32:18.008368Z","shell.execute_reply":"2022-03-06T20:32:19.693540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata = pd.read_csv(f'{BASE_DIR}/train_metadata.csv')\n\nwith open(f'{BASE_DIR}/scored_birds.json') as json_file:\n    scored_birds = json.load(json_file)","metadata":{"execution":{"iopub.status.busy":"2022-03-06T20:32:20.506264Z","iopub.execute_input":"2022-03-06T20:32:20.506849Z","iopub.status.idle":"2022-03-06T20:32:20.681752Z","shell.execute_reply.started":"2022-03-06T20:32:20.506809Z","shell.execute_reply":"2022-03-06T20:32:20.680387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"samples_n_channels = dict()\nsamples_sample_rate = dict()\nsamples_seconds = dict()\nsamples_max = dict()\n\nfor filename in tqdm(train_metadata['filename']):\n    frames, samples_sample_rate[filename] = torchaudio.load('../input/birdclef-2022/train_audio/' + filename)\n    samples_n_channels[filename] = frames.shape[0]\n    n_frames = frames.shape[1]\n    samples_seconds[filename] = float(n_frames) / float(samples_sample_rate[filename])\n    samples_max[filename] = torch.max(torch.abs(frames)).detach().numpy().item()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:31:05.227178Z","iopub.execute_input":"2022-03-03T20:31:05.227762Z","iopub.status.idle":"2022-03-03T20:31:14.732093Z","shell.execute_reply.started":"2022-03-03T20:31:05.22773Z","shell.execute_reply":"2022-03-03T20:31:14.731237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analize the audio files","metadata":{}},{"cell_type":"markdown","source":"First, for every bird in the scored bird, let's have a look on 60 seconds from a files","metadata":{}},{"cell_type":"code","source":"from glob import glob\nfor bird in scored_birds:\n    files = glob(f'../input/birdclef-2022/train_audio/{bird}/*.ogg')\n    filename = files[0]\n    frames, sample_rate = torchaudio.load(filename)\n    frames = frames.cpu().detach().numpy()\n    n_channels = frames.shape[0]\n    if frames.shape[1] > 60*sample_rate:\n        frames = frames[:, :60*sample_rate]\n        \n    plt.figure(figsize=(20, 3))\n    for c in range(n_channels):\n        plt.subplot(1, 2, c+1)\n        plt.plot(frames[c, :])\n        plt.title(f'{filename} channel {c}')\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-03-06T20:49:20.848531Z","iopub.execute_input":"2022-03-06T20:49:20.848803Z","iopub.status.idle":"2022-03-06T20:49:32.858638Z","shell.execute_reply.started":"2022-03-06T20:49:20.848776Z","shell.execute_reply":"2022-03-06T20:49:32.857845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### audio length distrubution","metadata":{}},{"cell_type":"code","source":"seconds = samples_seconds.values()\nprint(f'minimum length = {np.min(list(seconds))}')\nprint(f'maximum length = {np.max(list(seconds))}')\nprint(f'average length = {np.mean(list(seconds))}')\nn_bins = int(np.max(list(seconds)) / 5)\nplt.figure(figsize=(90, 20))\nplt.hist(list(seconds), bins=n_bins)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:31:16.169309Z","iopub.execute_input":"2022-03-03T20:31:16.169652Z","iopub.status.idle":"2022-03-03T20:31:16.802706Z","shell.execute_reply.started":"2022-03-03T20:31:16.169608Z","shell.execute_reply":"2022-03-03T20:31:16.801767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Number of channels in each audio file","metadata":{}},{"cell_type":"code","source":"print(f'The number of files with only one channel is {list(samples_n_channels.values()).count(1)}')\nprint(f'The number of files with only one channel is {list(samples_n_channels.values()).count(2)}')\n","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:31:16.803929Z","iopub.execute_input":"2022-03-03T20:31:16.805304Z","iopub.status.idle":"2022-03-03T20:31:16.811944Z","shell.execute_reply.started":"2022-03-03T20:31:16.805258Z","shell.execute_reply":"2022-03-03T20:31:16.81074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### sample rate","metadata":{}},{"cell_type":"code","source":"sample_rates = np.unique(list(samples_sample_rate.values()))\nprint(sample_rates)","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:31:16.813236Z","iopub.execute_input":"2022-03-03T20:31:16.813479Z","iopub.status.idle":"2022-03-03T20:31:16.829185Z","shell.execute_reply.started":"2022-03-03T20:31:16.813449Z","shell.execute_reply":"2022-03-03T20:31:16.828237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Maximum value in the audio files","metadata":{}},{"cell_type":"code","source":"plt.figure()\nmax_values = list(samples_max.values())\nplt.hist(max_values, bins=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-03T20:32:22.62234Z","iopub.execute_input":"2022-03-03T20:32:22.623047Z","iopub.status.idle":"2022-03-03T20:32:22.838376Z","shell.execute_reply.started":"2022-03-03T20:32:22.623004Z","shell.execute_reply":"2022-03-03T20:32:22.837757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analize the metadata","metadata":{}},{"cell_type":"markdown","source":"First let's try to look at all the metadat files (train_metadat.csv and eBird_Taxonomy_v2021.csv). This part is partially copied from https://www.kaggle.com/hasanbasriakcay/birdclef22-eda-noise-reduction)","metadata":{}},{"cell_type":"code","source":"train_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-06T21:12:21.383635Z","iopub.execute_input":"2022-03-06T21:12:21.383883Z","iopub.status.idle":"2022-03-06T21:12:21.409474Z","shell.execute_reply.started":"2022-03-06T21:12:21.383857Z","shell.execute_reply":"2022-03-06T21:12:21.408883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make sure that there are not nans in the metadata\ntrain_metadata.isna().sum()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now I want to check how much time every bird is a primary bird and how much time every bird appears to be scondary bird","metadata":{}},{"cell_type":"code","source":"train_metadata = pd.read_csv(f'{BASE_DIR}/train_metadata.csv')\n\nbirds_count_dict = train_metadata['primary_label'].value_counts().to_dict()\n\nsecondary_labels_as_arr = train_metadata['secondary_labels'].map(lambda x: x[1:-1].replace(\"'\", \"\").split(\", \"))\nall_secondary = secondary_labels_as_arr.aggregate('sum')\nbirds_count = [{'Bird': k, 'n_primary': birds_count_dict[k], 'n_secondary': all_secondary.count(k), 'total_time_primary': 0, 'total_time_secondary': 0} for k in birds_count_dict.keys()]\nbirds_count = pd.DataFrame(birds_count)\nbirds_count = birds_count.set_index('Bird', drop=True)\nfor _, row in train_metadata.iterrows():\n    seconds = samples_seconds[row.filename]\n    birds_count.loc[row.primary_label].total_time_primary += seconds\n    if len(row.secondary_labels) > 2:\n        arr = row.secondary_labels[1:-1].replace(\"'\", \"\").split(\", \")\n        for secondary_bird in arr:\n            birds_count.loc[secondary_bird].total_time_secondary += seconds\n\nbirds_count = birds_count.sort_values(by=['total_time_primary', 'total_time_secondary', 'n_primary', 'n_secondary'], ascending=False)\nprint(birds_count)\nbirds_count.to_csv('birds_count.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The full output is at birds_count.csv. As you can easily see, the data is extremely not balanced and we have to think about ways to make it balanced.\n\nLet's check the data only on the scored birds","metadata":{}},{"cell_type":"code","source":"n_primary = [birds_count.n_primary[bird] for bird in scored_birds]\nn_secondary = [birds_count.n_secondary[bird] for bird in scored_birds]\ntotal_time_primary = [birds_count.total_time_primary[bird] for bird in scored_birds]\ntotal_time_secondary = [birds_count.total_time_secondary[bird] for bird in scored_birds]\nprint(scored_birds)\nprint(n_primary)\nprint(n_secondary)\nprint(total_time_primary)\nprint(total_time_secondary)\n\nplt.figure(figsize=(20, 3))\nplt.bar(scored_birds, n_primary)\nplt.title('n_primary')\nplt.figure(figsize=(20, 3))\nplt.bar(scored_birds, n_secondary)\nplt.title('n_secondary')\nplt.figure(figsize=(20, 3))\nplt.bar(scored_birds, total_time_primary)\nplt.title('total_time_primary')\nplt.figure(figsize=(20, 3))\nplt.bar(scored_birds, total_time_secondary)\nplt.title('total_time_secondary')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's check the rating distribution.","metadata":{}},{"cell_type":"code","source":"plt.figure()\nplt.hist(train_metadata.rating, bins=10)\nplt.title('rating of all audio files')\n\nplt.figure()\nis_scored = train_metadata.primary_label.map(lambda x: x in scored_birds)\nrelevant_metadata = train_metadata[is_scored]\nplt.hist(relevant_metadata.rating, bins=10)\nplt.title('rating of the scored audio files')\n\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-06T21:15:39.143124Z","iopub.execute_input":"2022-03-06T21:15:39.143738Z","iopub.status.idle":"2022-03-06T21:15:39.530988Z","shell.execute_reply.started":"2022-03-06T21:15:39.143700Z","shell.execute_reply":"2022-03-06T21:15:39.530379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It looks like most of the samples have good quality.","metadata":{}}]}