{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Published on April 03, 2024. By Marília Prata, mpwolke","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.graph_objs as go\nimport plotly.offline as py\nimport plotly.express as px\n\n#Ignore warnings\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-04T01:27:29.343547Z","iopub.execute_input":"2024-04-04T01:27:29.343941Z","iopub.status.idle":"2024-04-04T01:27:42.218478Z","shell.execute_reply.started":"2024-04-04T01:27:29.343908Z","shell.execute_reply":"2024-04-04T01:27:42.217527Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Competition Citation:\n\n@misc{birdclef-2024,\n\n    author = {HCL-Rantig, Holger Klinck, Maggie, Sohier Dane, Stefan Kahl, Tom Denton},\n    \n    title = {BirdCLEF 2024},\n    publisher = {Kaggle},\n    year = {2024},\n    url = {https://kaggle.com/competitions/birdclef-2024}\n}","metadata":{}},{"cell_type":"markdown","source":"\"For this competition, you'll use your machine-learning skills to identify under-studied Indian bird species by sound. Specifically, you'll develop computational solutions to process continuous audio data and recognize the species by their calls. The best entries will be able to train reliable classifiers with limited training data. If successful, you'll help advance ongoing efforts to protect avian biodiversity in the Western Ghats, India, including those led by V. V. Robin's Lab at IISER Tirupati.\"\n\nhttps://www.kaggle.com/competitions/birdclef-2024/overview","metadata":{}},{"cell_type":"markdown","source":"![](https://pbs.twimg.com/media/EbHFD3JVAAA3Khf.jpg)https://twitter.com/moefcc/status/1275016847776088070","metadata":{}},{"cell_type":"markdown","source":"#eBird India\n\nThe Western Ghats endemic birds are more imperiled than previously thought.\n\nBy Vijay Ramesh July/13/2017\n \n\"Many species of birds in the Western Ghats have exclusively adapted to inhabit isolated pockets of this region and are found nowhere else in the world. These species are threatened as a result of widespread development and are in need of protection. However, their survival is currently being undermined by inaccurate range maps which quantify the size of species’ geographic spread and are a key determinant used by the International Union for Conservation of Nature (IUCN) to assign an appropriate threat status.\"\n\n\"While the environmental impacts of extensive deforestation, compounded by the intensifying effects of climate change, paint a bleak picture for the endemic birds of the Western Ghats, appropriate management – if implemented swiftly – could save these magnificent species. \"\n\nhttps://evolecol.weebly.com/blog/2017-western-ghats-birds\n\nIUCN greatly underestimates threat levels of endemic birds in the Western Ghats\n\nhttps://www.sciencedirect.com/science/article/abs/pii/S0006320716310588","metadata":{}},{"cell_type":"markdown","source":"#That Pew Pew Script is so cool that I had to split it into 2 Kaggle Notebooks\n\nPaulo Junqueira https://www.kaggle.com/code/paulojunqueira/pew-pew-overview-birdclef-2023/notebook","metadata":{}},{"cell_type":"code","source":"#By Paulo Junqueira https://www.kaggle.com/code/paulojunqueira/pew-pew-overview-birdclef-2023/notebook\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport re\nimport librosa\nimport librosa.display\n\nimport IPython.display as ipd\nfrom urllib.request import urlopen\nfrom datetime import datetime, timedelta\n\nimport plotly.graph_objects as go\nfrom scipy.interpolate import interp1d \nfrom bs4 import BeautifulSoup as bs\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\n# import noisereduce as nr\n\nfrom tqdm.notebook import tqdm\n# Pytorch\nimport torch\nimport torchaudio\nimport requests\nfrom PIL import Image\n\ndef get_link(url, name):\n    res = requests.get(url)\n    soup = bs(res.content)\n    external_link = soup.find_all(['a'], href = True, text = name)\n    url_2 = re.findall(r'\\\".*?\\\"', str(external_link[0]))[0].replace('\"', '')\n    return url_2\ndef wiki_link(url, name):\n    res = requests.get(url_3)\n    soup = bs(res.content)\n    external_link = soup.find_all(\"img\", src=re.compile(name))\n    \n    img = 'https:' +  find_between(str(external_link),'src=', ' ').replace('\"', '')\n    return img\n\ndef find_between( s, first, last ):\n    try:\n        start = s.index( first ) + len( first )\n        end = s.index( last, start )\n        return s[start:end]\n    except ValueError:\n        return \"\"\n    \ndef get_text(url, len_text):\n    res = requests.get(url)\n    soup = bs(res.content)\n    text = ''\n    for paragraph in soup.find_all('p'):\n        text += paragraph.text\n        \n    return text.split('\\n')[1:len_text+1]\n    \n\ndef get_spectogram(path):\n    \n    \n    data, sample_rate = librosa.load(path)\n    \n    stft = librosa.stft(data, n_fft=CFG.n_fft, hop_length=CFG.hop_length)\n    spectrogram = np.abs(stft)\n    x = librosa.amplitude_to_db(spectrogram)\n    \n    \n    #mel spectogram\n    transfomer = torchaudio.transforms.MelSpectrogram(sample_rate = sample_rate,\n                                                     n_fft = CFG.n_fft, \n                                                     win_length = CFG.win_length,\n                                                     n_mels = CFG.n_mels,\n                                                     f_min = CFG.f_min,\n                                                     f_max = CFG.f_max ).double()\n\n\n    wave = torch.from_numpy(data.copy())\n    mel_spectrogram = transfomer(wave)\n    \n    #PCEN melspectogram\n    pcen_spectogram = librosa.pcen(np.array(mel_spectrogram) * (2 ** 31), \n                                  eps = 1e-6,\n                                  gain = 0.8,\n                                  power = 0.25,\n                                  bias = 10, \n                                  sr = sample_rate,\n                                  hop_length = CFG.hop_length)\n    \n    \n    fig, ax = plt.subplots(ncols = 2, nrows = 1, figsize = (18,5))\n    \n    librosa.display.specshow(x, sr=sample_rate, hop_length=CFG.hop_length,ax = ax[0])\n    ax[0].set_title(\"Spectrogram - STFT\")\n    librosa.display.waveshow(data,ax = ax[1])\n    plt.show()\n    \n    fig, ax1 = plt.subplots(ncols = 2, nrows = 1, figsize = (18,5))\n    ax1[0].imshow(librosa.amplitude_to_db(mel_spectrogram))\n    ax1[0].set_title(\"Melspectogram\")\n    ax1[1].imshow((pcen_spectogram))\n    ax1[1].set_title(\"PCEN-Melspectogram\")\n    \n    \n    plt.show()\n\nsns.set_style(\"darkgrid\", {\"grid.color\": \".6\", \"grid.linestyle\": \":\"})","metadata":{"execution":{"iopub.status.busy":"2024-04-04T01:28:58.105188Z","iopub.execute_input":"2024-04-04T01:28:58.105914Z","iopub.status.idle":"2024-04-04T01:29:02.849415Z","shell.execute_reply.started":"2024-04-04T01:28:58.105876Z","shell.execute_reply":"2024-04-04T01:29:02.848445Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Paulo Junqueira https://www.kaggle.com/code/paulojunqueira/pew-pew-overview-birdclef-2023/notebook\n\nclass CFG():\n    '''Configuration File'''\n    n_fft = 2048\n    frame_size = 5 # seg\n    frame_step = 5  # seg\n\n    hop_length  = 128\n    frame_size_t  = 256\n    n_mels     = 250\n    win_length = 1024\n    f_min      = 500\n    f_max      = 9000","metadata":{"execution":{"iopub.status.busy":"2024-04-04T01:29:11.448509Z","iopub.execute_input":"2024-04-04T01:29:11.449544Z","iopub.status.idle":"2024-04-04T01:29:11.456184Z","shell.execute_reply.started":"2024-04-04T01:29:11.449502Z","shell.execute_reply":"2024-04-04T01:29:11.454677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Metadata file","metadata":{}},{"cell_type":"code","source":"meta = pd.read_csv('../input/birdclef-2024/train_metadata.csv')\nmeta['secondary_labels'] = meta['secondary_labels'].apply(lambda x: re.findall(r\"'(\\w+)'\", x))\nmeta['len_sec_labels'] = meta['secondary_labels'].map(len)\nmeta.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T01:29:52.857223Z","iopub.execute_input":"2024-04-04T01:29:52.858492Z","iopub.status.idle":"2024-04-04T01:29:53.440077Z","shell.execute_reply.started":"2024-04-04T01:29:52.858446Z","shell.execute_reply":"2024-04-04T01:29:53.438523Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"values = 20\n\ntop = meta['primary_label'].value_counts().head(values).index\nbotton = meta['primary_label'].value_counts().tail(values).index","metadata":{"execution":{"iopub.status.busy":"2024-04-04T01:29:59.164363Z","iopub.execute_input":"2024-04-04T01:29:59.165686Z","iopub.status.idle":"2024-04-04T01:29:59.180637Z","shell.execute_reply.started":"2024-04-04T01:29:59.165634Z","shell.execute_reply":"2024-04-04T01:29:59.179372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#By Paulo Junqueira https://www.kaggle.com/code/paulojunqueira/pew-pew-overview-birdclef-2023/notebook\n\nbirds_to_get_info = 20\n\nmapping = meta.groupby('primary_label').agg({'common_name':'unique'}).to_dict()['common_name']\n\nfor i, name in zip(tqdm(top[:birds_to_get_info]), list(map(lambda x: mapping[x][0], top))):\n    print('#'*200,'\\n',i, '->', name,'\\n','#'*200,'\\n')\n    sp1 = meta[meta['primary_label'] == i].sample(1,random_state = 666)\n    \n    url = sp1['url'].values[0]\n\n    path = sp1['filename'].values[0]\n    path ='/kaggle/input/birdclef-2024/train_audio/' + path\n    \n    try: \n        \n        url_2 = get_link(url, name)\n        url_3 = get_link(url_2, 'Wikipedia')\n        if i != 'barswa':\n            url_img = wiki_link(url_3, name.split(' ')[1] )\n        else:\n            #hardcoded\n            url_img = 'https://upload.wikimedia.org/wikipedia/commons/thumb/2/24/Landsvale.jpg/220px-Landsvale.jpg'\n        \n        \n        \n        \n        img = Image.open(urlopen(url_img))\n        print('> Bird Image ***')\n        plt.imshow(img)\n        plt.show()\n    except:\n        print('No Image Found')\n    \n    info = get_text(url_3, 3)\n    print('> Wikipedia Info ***')\n    print('\\n'.join(info))\n    print('External Links:\\n', url,'\\n', url_2,'\\n', url_3)\n    #sr = 22050 # sample rate\n    print('\\n> Bird Sound ***')\n    ipd.display(ipd.Audio(path))\n    \n    #Calculate Spectogram\n    print('\\n> Bird Spectogram ***')\n    get_spectogram(path)\n    print('\\n')","metadata":{"execution":{"iopub.status.busy":"2024-04-04T01:30:05.458533Z","iopub.execute_input":"2024-04-04T01:30:05.458951Z","iopub.status.idle":"2024-04-04T01:34:54.297240Z","shell.execute_reply.started":"2024-04-04T01:30:05.458917Z","shell.execute_reply":"2024-04-04T01:34:54.295815Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Acknowledgements:\n\nPaulo Junqueira https://www.kaggle.com/code/paulojunqueira/pew-pew-overview-birdclef-2023/notebook","metadata":{}}]}