{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport re\nimport librosa\nimport librosa.display\n\nimport IPython.display as ipd\nfrom urllib.request import urlopen\nfrom datetime import datetime, timedelta\n\nimport plotly.graph_objects as go\nfrom scipy.interpolate import interp1d \nfrom bs4 import BeautifulSoup as bs\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\n# import noisereduce as nr\n\nfrom tqdm.notebook import tqdm\n# Pytorch\nimport torch\nimport torchaudio\nimport requests\nfrom PIL import Image\n\ndef get_link(url, name):\n    res = requests.get(url)\n    soup = bs(res.content)\n    external_link = soup.find_all(['a'], href = True, text = name)\n    url_2 = re.findall(r'\\\".*?\\\"', str(external_link[0]))[0].replace('\"', '')\n    return url_2\ndef wiki_link(url, name):\n    res = requests.get(url_3)\n    soup = bs(res.content)\n    external_link = soup.find_all(\"img\", src=re.compile(name))\n    \n    img = 'https:' +  find_between(str(external_link),'src=', ' ').replace('\"', '')\n    return img\n\ndef find_between( s, first, last ):\n    try:\n        start = s.index( first ) + len( first )\n        end = s.index( last, start )\n        return s[start:end]\n    except ValueError:\n        return \"\"\n    \ndef get_text(url, len_text):\n    res = requests.get(url)\n    soup = bs(res.content)\n    text = ''\n    for paragraph in soup.find_all('p'):\n        text += paragraph.text\n        \n    return text.split('\\n')[1:len_text+1]\n    \n\ndef get_spectogram(path):\n    \n    \n    data, sample_rate = librosa.load(path)\n    \n    stft = librosa.stft(data, n_fft=CFG.n_fft, hop_length=CFG.hop_length)\n    spectrogram = np.abs(stft)\n    x = librosa.amplitude_to_db(spectrogram)\n    \n    \n    #mel spectogram\n    transfomer = torchaudio.transforms.MelSpectrogram(sample_rate = sample_rate,\n                                                     n_fft = CFG.n_fft, \n                                                     win_length = CFG.win_length,\n                                                     n_mels = CFG.n_mels,\n                                                     f_min = CFG.f_min,\n                                                     f_max = CFG.f_max ).double()\n\n\n    wave = torch.from_numpy(data.copy())\n    mel_spectrogram = transfomer(wave)\n    \n    #PCEN melspectogram\n    pcen_spectogram = librosa.pcen(np.array(mel_spectrogram) * (2 ** 31), \n                                  eps = 1e-6,\n                                  gain = 0.8,\n                                  power = 0.25,\n                                  bias = 10, \n                                  sr = sample_rate,\n                                  hop_length = CFG.hop_length)\n    \n    \n    fig, ax = plt.subplots(ncols = 2, nrows = 1, figsize = (18,5))\n    \n    librosa.display.specshow(x, sr=sample_rate, hop_length=CFG.hop_length,ax = ax[0])\n    ax[0].set_title(\"Spectrogram - STFT\")\n    librosa.display.waveshow(data,ax = ax[1])\n    plt.show()\n    \n    fig, ax1 = plt.subplots(ncols = 2, nrows = 1, figsize = (18,5))\n    ax1[0].imshow(librosa.amplitude_to_db(mel_spectrogram))\n    ax1[0].set_title(\"Melspectogram\")\n    ax1[1].imshow((pcen_spectogram))\n    ax1[1].set_title(\"PCEN-Melspectogram\")\n    \n    \n    plt.show()\n\nsns.set_style(\"darkgrid\", {\"grid.color\": \".6\", \"grid.linestyle\": \":\"})","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-31T17:29:07.164935Z","iopub.execute_input":"2023-03-31T17:29:07.165467Z","iopub.status.idle":"2023-03-31T17:29:12.576859Z","shell.execute_reply.started":"2023-03-31T17:29:07.165423Z","shell.execute_reply":"2023-03-31T17:29:12.575111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG():\n    '''Configuration File'''\n    n_fft = 2048\n    frame_size = 5 # seg\n    frame_step = 5  # seg\n\n    hop_length  = 128\n    frame_size_t  = 256\n    n_mels     = 250\n    win_length = 1024\n    f_min      = 500\n    f_max      = 9000\n","metadata":{"execution":{"iopub.status.busy":"2023-03-31T17:29:12.579360Z","iopub.execute_input":"2023-03-31T17:29:12.580236Z","iopub.status.idle":"2023-03-31T17:29:12.588894Z","shell.execute_reply.started":"2023-03-31T17:29:12.580187Z","shell.execute_reply":"2023-03-31T17:29:12.587346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"00\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Data Overview </p>\n<div style=\"font-family: courier; font-size:18px\">\n    \n<li> General Overview of data\n<li> Top Birds\n<li> Secondary Labels\n<li> Spectograms\n<li> <b>!!!Under Developement!!!<b>","metadata":{}},{"cell_type":"code","source":"meta = pd.read_csv('../input/birdclef-2023/train_metadata.csv')\nmeta['secondary_labels'] = meta['secondary_labels'].apply(lambda x: re.findall(r\"'(\\w+)'\", x))\nmeta['len_sec_labels'] = meta['secondary_labels'].map(len)\nmeta.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-31T17:29:12.591118Z","iopub.execute_input":"2023-03-31T17:29:12.591599Z","iopub.status.idle":"2023-03-31T17:29:12.809976Z","shell.execute_reply.started":"2023-03-31T17:29:12.591553Z","shell.execute_reply":"2023-03-31T17:29:12.808797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta.info()","metadata":{"execution":{"iopub.status.busy":"2023-03-31T17:29:12.813253Z","iopub.execute_input":"2023-03-31T17:29:12.813824Z","iopub.status.idle":"2023-03-31T17:29:12.846915Z","shell.execute_reply.started":"2023-03-31T17:29:12.813773Z","shell.execute_reply":"2023-03-31T17:29:12.845042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-31T17:29:12.848534Z","iopub.execute_input":"2023-03-31T17:29:12.849846Z","iopub.status.idle":"2023-03-31T17:29:12.859088Z","shell.execute_reply.started":"2023-03-31T17:29:12.849790Z","shell.execute_reply":"2023-03-31T17:29:12.857595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(nrows = 1, ncols = 2, figsize = (18,8),gridspec_kw={'width_ratios': [2, 1]})\nplot_temp = meta['len_sec_labels'].value_counts().reset_index()\nsns.barplot(x = 'index', y = 'len_sec_labels',data =plot_temp, ax = ax[0]  )\nsns.barplot(x = 'index', y = 'len_sec_labels',data =plot_temp.query('index > 3'), ax = ax[1]  )\n\n\nax[0].set_title(f'Total Number of Birds Sounds on Background', fontdict = {'fontsize':20})\nax[1].set_title(f'4 or more', fontdict = {'fontsize':20})\nax[0].set_xlabel('Number', fontdict = {'fontsize':16})\nax[1].set_xlabel('Number', fontdict = {'fontsize':16})\nax[0].set_ylabel('Count', fontdict = {'fontsize':16})\nax[1].set_ylabel('')","metadata":{"execution":{"iopub.status.busy":"2023-03-31T17:29:12.860441Z","iopub.execute_input":"2023-03-31T17:29:12.860811Z","iopub.status.idle":"2023-03-31T17:29:13.513747Z","shell.execute_reply.started":"2023-03-31T17:29:12.860775Z","shell.execute_reply":"2023-03-31T17:29:13.512399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = 20\nfig, ax = plt.subplots(nrows = 1, ncols = 1, figsize = (10,8))\nsns.barplot(y = 'index', x = 'secondary_labels',data = meta['secondary_labels'].explode().value_counts().head(values).reset_index(),ax = ax,palette = 'rocket')\nax.set_title(f'Top {values} Birds Found on Background as Noise', fontdict = {'fontsize':20})\nax.set_xlabel('Frequency', fontdict = {'fontsize':16})\nax.set_ylabel('Birds Common Name', fontdict = {'fontsize':16})\n\n\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-03-31T17:29:13.515407Z","iopub.execute_input":"2023-03-31T17:29:13.516329Z","iopub.status.idle":"2023-03-31T17:29:14.133355Z","shell.execute_reply.started":"2023-03-31T17:29:13.516275Z","shell.execute_reply":"2023-03-31T17:29:14.132353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = 20\n\ntop = meta['primary_label'].value_counts().head(values).index\nbotton = meta['primary_label'].value_counts().tail(values).index\n\nfig, ax = plt.subplots(nrows = 1, ncols = 2, figsize = (18,8))\nsns.barplot(y = 'index', x = 'common_name',data = meta['common_name'].value_counts().head(values).reset_index(),ax = ax[0],palette = 'rocket')\nsns.barplot(y = 'index', x = 'common_name',data = meta['common_name'].value_counts().tail(values).reset_index(),ax = ax[1], palette = 'viridis')\nax[0].set_title(f'Top {values} Birds in Primary Label', fontdict = {'fontsize':20})\nax[1].set_title(f'Botton {values} Birds, Primary Label', fontdict = {'fontsize':20})\nax[0].set_xlabel('Frequency', fontdict = {'fontsize':16})\nax[1].set_xlabel('Frequency', fontdict = {'fontsize':16})\nax[0].set_ylabel('Birds Common Name', fontdict = {'fontsize':16})\nax[1].set_ylabel('')\n\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-03-31T17:29:14.134718Z","iopub.execute_input":"2023-03-31T17:29:14.135296Z","iopub.status.idle":"2023-03-31T17:29:15.581353Z","shell.execute_reply.started":"2023-03-31T17:29:14.135258Z","shell.execute_reply":"2023-03-31T17:29:15.580266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_plot = meta.groupby(['primary_label','latitude', 'longitude']).count().reset_index()[['primary_label','scientific_name','latitude', 'longitude']].rename(columns = {'scientific_name':'count'})\nmeta_2 = meta.merge(df_plot, on = ['primary_label','latitude', 'longitude'], how = 'left').dropna(subset = ['count'])\nmeta_2['count'] = meta_2['count'].astype('int')\n\nvalues_list = meta_2['count'].values.tolist()\n\ninterpolation = interp1d([1, max(values_list)], [3,20])\nradius = interpolation(values_list)\nfig = go.Figure(go.Densitymapbox(lat =meta_2['latitude'],lon = meta_2['longitude'], radius = radius,z = meta_2['count']))\n\nfig.update_layout(mapbox_style=\"open-street-map\",height = 800,\n                  mapbox = {\n                          'center': {'lat': 0, \n                          'lon': 0},\n                      'zoom':0\n                  })\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-31T17:29:15.582759Z","iopub.execute_input":"2023-03-31T17:29:15.583451Z","iopub.status.idle":"2023-03-31T17:29:16.004628Z","shell.execute_reply.started":"2023-03-31T17:29:15.583408Z","shell.execute_reply":"2023-03-31T17:29:16.003363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"00\"></a>\n# <p style=\"background-color:#002663;height: 60px;text-align: center;vertical-align: middle;line-height: 60px;;font-family:courier;color:#FFFFFF;font-size:120%;text-align:center;border-radius:12px 12px;\"> Top Bird Train Date </p>\n<div style=\"font-family: courier; font-size:18px\"> \n<li> Bird info\n<li>Bird Image\n<li>Bird Song\n<li>Spectogram\n\n*Some infos were scrapped from wikipedia","metadata":{}},{"cell_type":"code","source":"birds_to_get_info = 20\n\nmapping = meta.groupby('primary_label').agg({'common_name':'unique'}).to_dict()['common_name']\n\nfor i, name in zip(tqdm(top[:birds_to_get_info]), list(map(lambda x: mapping[x][0], top))):\n    print('#'*200,'\\n',i, '->', name,'\\n','#'*200,'\\n')\n    sp1 = meta[meta['primary_label'] == i].sample(1,random_state = 666)\n    \n    url = sp1['url'].values[0]\n\n    path = sp1['filename'].values[0]\n    path ='/kaggle/input/birdclef-2023/train_audio/' + path\n    \n    try: \n        \n        url_2 = get_link(url, name)\n        url_3 = get_link(url_2, 'Wikipedia')\n        if i != 'barswa':\n            url_img = wiki_link(url_3, name.split(' ')[1] )\n        else:\n            #hardcoded\n            url_img = 'https://upload.wikimedia.org/wikipedia/commons/thumb/2/24/Landsvale.jpg/220px-Landsvale.jpg'\n        \n        \n        \n        \n        img = Image.open(urlopen(url_img))\n        print('> Bird Image ***')\n        plt.imshow(img)\n        plt.show()\n    except:\n        print('No Image Found')\n    \n    info = get_text(url_3, 3)\n    print('> Wikipedia Info ***')\n    print('\\n'.join(info))\n    print('External Links:\\n', url,'\\n', url_2,'\\n', url_3)\n\n    print('\\n> Bird Sound ***')\n    ipd.display(ipd.Audio(path))\n    \n    #Calculate Spectogram\n    print('\\n> Bird Spectogram ***')\n    get_spectogram(path)\n    print('\\n')\n","metadata":{"execution":{"iopub.status.busy":"2023-03-10T18:41:04.247646Z","iopub.execute_input":"2023-03-10T18:41:04.248315Z","iopub.status.idle":"2023-03-10T18:41:09.156090Z","shell.execute_reply.started":"2023-03-10T18:41:04.248251Z","shell.execute_reply":"2023-03-10T18:41:09.154765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Waves Sampling","metadata":{}},{"cell_type":"code","source":"path = '/kaggle/input/birdclef-2023/train_audio/'\nfiles = path + meta['filename'].sample(5, random_state = 5052023).values \n\nSR = 32000\nfor f in files:\n    data, _ = librosa.load(f)\n    m = data[:20*SR].max()\n    fig, ax = plt.subplots(figsize = (20, 7), ncols = 4 )\n    for idx, t in enumerate(range(0,20, 5)):\n        \n        ax[idx].plot(data[t*SR:(t+5)*SR])\n        ax[idx].set_ylim(-m, m)\n    print(f.split('/')[-2])\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-03-31T18:29:13.518883Z","iopub.execute_input":"2023-03-31T18:29:13.519298Z","iopub.status.idle":"2023-03-31T18:29:20.050358Z","shell.execute_reply.started":"2023-03-31T18:29:13.519260Z","shell.execute_reply":"2023-03-31T18:29:20.048986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}