{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n# #         print(os.path.join(dirname, filename))\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport re\nimport librosa\nimport librosa.display\n\nimport IPython.display as ipd\n\nfrom datetime import datetime, timedelta\n\nimport plotly.graph_objects as go\nfrom scipy.interpolate import interp1d \n\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\n# import noisereduce as nr\n\n# Pytorch\nimport torch\nimport torchaudio\n\n\nsns.set_style(\"darkgrid\", {\"grid.color\": \".6\", \"grid.linestyle\": \":\"})","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-03T19:08:09.003227Z","iopub.execute_input":"2022-05-03T19:08:09.004174Z","iopub.status.idle":"2022-05-03T19:08:10.107598Z","shell.execute_reply.started":"2022-05-03T19:08:09.004119Z","shell.execute_reply":"2022-05-03T19:08:10.106936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class configuraion():\n    '''Configuration File'''\n    n_fft = 2048\n    \n    ################## Spectogram\n    #image - Spect\n    frame_size = 5 # seg\n    frame_step = 5  # seg\n\n    # Spectogram Transformer\n    # Default Spec (librosa)\n    hop_length  = 128\n    frame_size_t  = 256\n\n    #Mel spect (Torch)\n    n_mels     = 250\n    win_length = 1024\n    f_min      = 500\n    f_max      = 9000\n        \n    \n    n_show_birds = 5\n    \nCFG = configuraion()","metadata":{"execution":{"iopub.status.busy":"2022-05-03T19:07:28.455725Z","iopub.execute_input":"2022-05-03T19:07:28.456048Z","iopub.status.idle":"2022-05-03T19:07:28.462678Z","shell.execute_reply.started":"2022-05-03T19:07:28.456014Z","shell.execute_reply":"2022-05-03T19:07:28.461684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# BirdCLEF 2022 Competition Warming up !\n\n**under construction**  \nThe main objective of this notebook is to warm up of for the BirdCLEF 2022 Competion!.Therefore, to understand the basics of the dataset and share!  \nTo achieve this goal, the following steps are implemented:\n- Load data and understand the meta data and traning audio data\n- EDA of meta data\n- Check some samples of audio and its´ spectograms","metadata":{}},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"markdown","source":"- primary_label: Label of main bird\\specie\n- secondary_label: Labels of bird sounds in background. If none, empty list\n- latitude\\longitude: geografic location of bird species\n","metadata":{}},{"cell_type":"code","source":"meta = pd.read_csv('../input/birdclef-2022/train_metadata.csv')\nmeta.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-05-03T18:58:18.137217Z","iopub.execute_input":"2022-05-03T18:58:18.137589Z","iopub.status.idle":"2022-05-03T18:58:18.294726Z","shell.execute_reply.started":"2022-05-03T18:58:18.137557Z","shell.execute_reply":"2022-05-03T18:58:18.293881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eBird = pd.read_csv('../input/birdclef-2022/eBird_Taxonomy_v2021.csv')\neBird.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-05-03T18:58:18.296634Z","iopub.execute_input":"2022-05-03T18:58:18.296888Z","iopub.status.idle":"2022-05-03T18:58:18.393287Z","shell.execute_reply.started":"2022-05-03T18:58:18.296858Z","shell.execute_reply":"2022-05-03T18:58:18.392369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"meta['secondary_labels'] = meta['secondary_labels'].apply(lambda x: re.findall(r\"'(\\w+)'\", x))","metadata":{"execution":{"iopub.status.busy":"2022-05-03T18:58:18.394832Z","iopub.execute_input":"2022-05-03T18:58:18.395127Z","iopub.status.idle":"2022-05-03T18:58:18.427619Z","shell.execute_reply.started":"2022-05-03T18:58:18.395083Z","shell.execute_reply":"2022-05-03T18:58:18.426967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 1, ncols = 2, figsize = (18,8),gridspec_kw={'width_ratios': [2, 1]})\nplot_temp = meta['secondary_labels'].apply(len).sort_values(ascending = False).reset_index()\nsns.countplot('secondary_labels',data = plot_temp ,palette = 'rocket', ax = ax[0])\nsns.countplot('secondary_labels',data = plot_temp.query('secondary_labels > 2'),palette = 'rocket', ax = ax[1])\n\nax[0].set_title(f'Total Number of Birds Sounds on Background', fontdict = {'fontsize':20})\nax[1].set_title(f'3 or more', fontdict = {'fontsize':20})\nax[0].set_xlabel('Number', fontdict = {'fontsize':16})\nax[1].set_xlabel('Number', fontdict = {'fontsize':16})\nax[0].set_ylabel('Count', fontdict = {'fontsize':16})\nax[1].set_ylabel('')","metadata":{"execution":{"iopub.status.busy":"2022-05-03T18:58:18.428736Z","iopub.execute_input":"2022-05-03T18:58:18.429168Z","iopub.status.idle":"2022-05-03T18:58:18.968077Z","shell.execute_reply.started":"2022-05-03T18:58:18.429113Z","shell.execute_reply":"2022-05-03T18:58:18.965752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = 20\nfig, ax = plt.subplots(nrows = 1, ncols = 1, figsize = (10,8))\nsns.barplot(y = 'index', x = 'secondary_labels',data = meta['secondary_labels'].explode().value_counts().head(values).reset_index(),ax = ax,palette = 'rocket')\nax.set_title(f'Top {values} Birds Found on Background as Noise', fontdict = {'fontsize':20})\nax.set_xlabel('Frequency', fontdict = {'fontsize':16})\nax.set_ylabel('Birds Common Name', fontdict = {'fontsize':16})\n\n\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-05-03T18:58:18.969206Z","iopub.execute_input":"2022-05-03T18:58:18.970008Z","iopub.status.idle":"2022-05-03T18:58:19.506453Z","shell.execute_reply.started":"2022-05-03T18:58:18.969968Z","shell.execute_reply":"2022-05-03T18:58:19.505410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = 20\n\ntop = meta['primary_label'].value_counts().head(values).index\nbotton = meta['primary_label'].value_counts().tail(values).index\n\nfig, ax = plt.subplots(nrows = 1, ncols = 2, figsize = (18,8))\nsns.barplot(y = 'index', x = 'common_name',data = meta['common_name'].value_counts().head(values).reset_index(),ax = ax[0],palette = 'rocket')\nsns.barplot(y = 'index', x = 'common_name',data = meta['common_name'].value_counts().tail(values).reset_index(),ax = ax[1], palette = 'viridis')\nax[0].set_title(f'Top {values} Birds', fontdict = {'fontsize':20})\nax[1].set_title(f'Botton {values} Birds', fontdict = {'fontsize':20})\nax[0].set_xlabel('Frequency', fontdict = {'fontsize':16})\nax[1].set_xlabel('Frequency', fontdict = {'fontsize':16})\nax[0].set_ylabel('Birds Common Name', fontdict = {'fontsize':16})\nax[1].set_ylabel('')\n\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-05-03T18:58:19.507601Z","iopub.execute_input":"2022-05-03T18:58:19.507881Z","iopub.status.idle":"2022-05-03T18:58:20.631207Z","shell.execute_reply.started":"2022-05-03T18:58:19.507841Z","shell.execute_reply":"2022-05-03T18:58:20.630454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Map","metadata":{}},{"cell_type":"code","source":"df_plot = meta.groupby(['primary_label','latitude', 'longitude']).count().reset_index()[['primary_label','scientific_name','latitude', 'longitude']].rename(columns = {'scientific_name':'count'})\nmeta = meta.merge(df_plot, on = ['primary_label','latitude', 'longitude'], how = 'left')\n\nvalues_list = meta['count'].values.tolist()\n\ninterpolation = interp1d([1, max(values_list)], [3,20])\nradius = interpolation(values_list)\nfig = go.Figure(go.Densitymapbox(lat =meta['latitude'],lon = meta['longitude'], radius = radius,z = meta['count']))\n\nfig.update_layout(mapbox_style=\"open-street-map\",height = 800,\n                  mapbox = {\n                          'center': {'lat': 0, \n                          'lon': 0},\n                      'zoom':0\n                  })\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-05-03T18:58:20.632266Z","iopub.execute_input":"2022-05-03T18:58:20.632947Z","iopub.status.idle":"2022-05-03T18:58:20.865310Z","shell.execute_reply.started":"2022-05-03T18:58:20.632909Z","shell.execute_reply":"2022-05-03T18:58:20.864571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Time ","metadata":{}},{"cell_type":"code","source":"def round_date(date, delta = 30, th = 10):\n    date = date.to_pydatetime()\n    x = date.minute\n    if ((x >= (delta - th)) & (x < delta)) or (x > (delta + th)):\n#         print('Up')\n        date = date + (datetime.min - date) % timedelta(minutes = delta)\n    elif ((x <= (delta+ th )) & (x > delta)) or (x < (delta - th)):\n#         print('down')\n        date = date - (date - datetime.min) % timedelta(minutes = delta)\n\n    \n    return date.time().strftime(\"%H:%M\")\n\nmeta['time_tf']  = pd.to_datetime(meta['time'], errors = 'coerce').dropna().apply(lambda x:round_date(x))\nmeta.dropna(subset=['time_tf'], inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-05-03T18:58:20.868484Z","iopub.execute_input":"2022-05-03T18:58:20.869092Z","iopub.status.idle":"2022-05-03T18:58:21.140507Z","shell.execute_reply.started":"2022-05-03T18:58:20.869042Z","shell.execute_reply":"2022-05-03T18:58:21.139680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (15,8))\nsns.countplot(x = 'time_tf', data = meta.sort_values(by = 'time_tf'), palette = 'magma')\nplt.xticks(rotation=45)\nplt.xlabel('Time', fontdict = {'fontsize':18})\nplt.ylabel('Frequency', fontdict = {'fontsize':18})\nplt.title('Birds Registers` Time',fontdict = {'fontsize':18})\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2022-05-03T18:58:21.142239Z","iopub.execute_input":"2022-05-03T18:58:21.142565Z","iopub.status.idle":"2022-05-03T18:58:22.327206Z","shell.execute_reply.started":"2022-05-03T18:58:21.142511Z","shell.execute_reply":"2022-05-03T18:58:22.326259Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Audio Sample","metadata":{}},{"cell_type":"markdown","source":"## Top Frequent Birds","metadata":{}},{"cell_type":"code","source":"for i in top[:CFG.n_show_birds]:\n    print(i)\n    path = meta[meta['primary_label'] == i].sample(1,random_state = 666)['filename'].values[0]\n    path ='/kaggle/input/birdclef-2022/train_audio/' + path\n    ipd.display(ipd.Audio(path))\n    ","metadata":{"execution":{"iopub.status.busy":"2022-05-03T18:58:22.328922Z","iopub.execute_input":"2022-05-03T18:58:22.329239Z","iopub.status.idle":"2022-05-03T18:58:22.472147Z","shell.execute_reply.started":"2022-05-03T18:58:22.329196Z","shell.execute_reply":"2022-05-03T18:58:22.471206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in top[:CFG.n_show_birds]:\n    print(i)\n    \n    path = meta[meta['primary_label'] == i].sample(1,random_state = 666)['filename'].values[0]\n    path ='/kaggle/input/birdclef-2022/train_audio/' + path\n    \n    data, sample_rate = librosa.load(path)\n    \n    stft = librosa.stft(data, n_fft=CFG.n_fft, hop_length=CFG.hop_length)\n    spectrogram = np.abs(stft)\n    x = librosa.amplitude_to_db(spectrogram)\n    \n    \n    #mel spectogram\n    transfomer = torchaudio.transforms.MelSpectrogram(sample_rate = sample_rate,\n                                                     n_fft = CFG.n_fft, \n                                                     win_length = CFG.win_length,\n                                                     n_mels = CFG.n_mels,\n                                                     f_min = CFG.f_min,\n                                                     f_max = CFG.f_max ).double()\n\n\n    wave = torch.from_numpy(data.copy())\n    mel_spectrogram = transfomer(wave)\n    \n    #PCEN melspectogram\n    pcen_spectogram = librosa.pcen(np.array(mel_spectrogram) * (2 ** 31), \n                                  eps = 1e-6,\n                                  gain = 0.8,\n                                  power = 0.25,\n                                  bias = 10, \n                                  sr = sample_rate,\n                                  hop_length = CFG.hop_length)\n    \n    \n    fig, ax = plt.subplots(ncols = 2, nrows = 1, figsize = (12,5))\n    \n    librosa.display.specshow(x, sr=sample_rate, hop_length=CFG.hop_length,ax = ax[0])\n    ax[0].set_title(\"Spectrogram - STFT\")\n    librosa.display.waveshow(data,ax = ax[1])\n    plt.show()\n    \n    fig, ax1 = plt.subplots(ncols = 2, nrows = 1, figsize = (18,5))\n    ax1[0].imshow(librosa.amplitude_to_db(mel_spectrogram))\n    ax1[0].set_title(\"Melspectogram\")\n    ax1[1].imshow((pcen_spectogram))\n    ax1[1].set_title(\"PCEN-Melspectogram\")\n    \n    \n    plt.show()\n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-05-03T19:16:43.816280Z","iopub.execute_input":"2022-05-03T19:16:43.816600Z","iopub.status.idle":"2022-05-03T19:17:28.988028Z","shell.execute_reply.started":"2022-05-03T19:16:43.816565Z","shell.execute_reply":"2022-05-03T19:17:28.987417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Botton Frequent Birds","metadata":{}},{"cell_type":"code","source":"for i in botton[:CFG.n_show_birds]:\n    print(i)\n    path = meta[meta['primary_label'] == i].sample(1,random_state = 666)['filename'].values[0]\n    path ='/kaggle/input/birdclef-2022/train_audio/' + path\n    ipd.display(ipd.Audio(path))\n    ","metadata":{"execution":{"iopub.status.busy":"2022-05-03T18:58:39.385522Z","iopub.execute_input":"2022-05-03T18:58:39.386347Z","iopub.status.idle":"2022-05-03T18:58:39.547632Z","shell.execute_reply.started":"2022-05-03T18:58:39.386299Z","shell.execute_reply":"2022-05-03T18:58:39.547071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in botton[:CFG.n_show_birds]:\n    print(i)\n    \n    path = meta[meta['primary_label'] == i].sample(1,random_state = 666)['filename'].values[0]\n    path ='/kaggle/input/birdclef-2022/train_audio/' + path\n    \n    data, sample_rate = librosa.load(path)\n    \n    stft = librosa.stft(data, n_fft=CFG.n_fft, hop_length=CFG.hop_length)\n    spectrogram = np.abs(stft)\n    x = librosa.amplitude_to_db(spectrogram)\n    \n    \n    #mel spectogram\n    transfomer = torchaudio.transforms.MelSpectrogram(sample_rate = sample_rate,\n                                                     n_fft = CFG.n_fft, \n                                                     win_length = CFG.win_length,\n                                                     n_mels = CFG.n_mels,\n                                                     f_min = CFG.f_min,\n                                                     f_max = CFG.f_max ).double()\n\n\n    wave = torch.from_numpy(data.copy())\n    mel_spectrogram = transfomer(wave)\n    \n    #PCEN melspectogram\n    pcen_spectogram = librosa.pcen(np.array(mel_spectrogram) * (2 ** 31), \n                                  eps = 1e-6,\n                                  gain = 0.8,\n                                  power = 0.25,\n                                  bias = 10, \n                                  sr = sample_rate,\n                                  hop_length = CFG.hop_length)\n    \n    \n    fig, ax = plt.subplots(ncols = 2, nrows = 1, figsize = (12,5))\n    \n    librosa.display.specshow(x, sr=sample_rate, hop_length=CFG.hop_length,ax = ax[0])\n    ax[0].set_title(\"Spectrogram - STFT\")\n    librosa.display.waveshow(data,ax = ax[1])\n    plt.show()\n    \n    fig, ax1 = plt.subplots(ncols = 2, nrows = 1, figsize = (18,5))\n    ax1[0].imshow(librosa.amplitude_to_db(mel_spectrogram))\n    ax1[0].set_title(\"Melspectogram\")\n    ax1[1].imshow((pcen_spectogram))\n    ax1[1].set_title(\"PCEN-Melspectogram\")\n    \n    \n    plt.show()\n    ","metadata":{"execution":{"iopub.status.busy":"2022-05-03T19:17:38.914563Z","iopub.execute_input":"2022-05-03T19:17:38.915253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}