{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n    #for filename in filenames:\n        #print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:20:57.378398Z","iopub.execute_input":"2025-03-12T06:20:57.378994Z","iopub.status.idle":"2025-03-12T06:20:57.837890Z","shell.execute_reply.started":"2025-03-12T06:20:57.378941Z","shell.execute_reply":"2025-03-12T06:20:57.836819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\nimport librosa\nimport librosa.display\nimport matplotlib.pyplot as plt\nimport IPython.display as ipd\nfrom urllib.request import urlopen\nfrom datetime import datetime, timedelta\n\nimport plotly.graph_objects as go\nfrom scipy.interpolate import interp1d \nfrom bs4 import BeautifulSoup as bs\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\n# import noisereduce as nr\n\nfrom tqdm.notebook import tqdm\n# Pytorch\nimport torch\nimport torchaudio\nimport requests\nfrom PIL import Image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:20:57.839510Z","iopub.execute_input":"2025-03-12T06:20:57.840033Z","iopub.status.idle":"2025-03-12T06:21:00.608917Z","shell.execute_reply.started":"2025-03-12T06:20:57.840001Z","shell.execute_reply":"2025-03-12T06:21:00.607777Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta = pd.read_csv('../input/birdclef-2025/train.csv')\nmeta['secondary_labels'] = meta['secondary_labels'].apply(lambda x: re.findall(r\"'(\\w+)'\", x))\nmeta['len_sec_labels'] = meta['secondary_labels'].map(len)\nmeta.head(2)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:21:00.610759Z","iopub.execute_input":"2025-03-12T06:21:00.611364Z","iopub.status.idle":"2025-03-12T06:21:00.924659Z","shell.execute_reply.started":"2025-03-12T06:21:00.611332Z","shell.execute_reply":"2025-03-12T06:21:00.923515Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#By Paulo Junqueira https://www.kaggle.com/code/paulojunqueira/pew-pew-overview-birdclef-2023/notebook\n\n#Two lines Required to Plot Plotly\nimport plotly.io as pio\npio.renderers.default = 'iframe'\n\n\ndf_plot = meta.groupby(['primary_label','latitude', 'longitude']).count().reset_index()[['primary_label','scientific_name','latitude', 'longitude']].rename(columns = {'scientific_name':'count'})\nmeta_2 = meta.merge(df_plot, on = ['primary_label','latitude', 'longitude'], how = 'left').dropna(subset = ['count'])\nmeta_2['count'] = meta_2['count'].astype('int')\n\nvalues_list = meta_2['count'].values.tolist()\n\ninterpolation = interp1d([1, max(values_list)], [3,20])\nradius = interpolation(values_list)\nfig = go.Figure(go.Densitymapbox(lat =meta_2['latitude'],lon = meta_2['longitude'], radius = radius,z = meta_2['count']))\n\nfig.update_layout(mapbox_style=\"open-street-map\",height = 800,\n                  mapbox = {\n                          'center': {'lat': 0, \n                          'lon': 0},\n                      'zoom':0\n                  })\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:21:00.926536Z","iopub.execute_input":"2025-03-12T06:21:00.926905Z","iopub.status.idle":"2025-03-12T06:21:01.269534Z","shell.execute_reply.started":"2025-03-12T06:21:00.926872Z","shell.execute_reply":"2025-03-12T06:21:01.268617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"taxonomy = pd.read_csv(\"/kaggle/input/birdclef-2025/taxonomy.csv\")\npd.set_option('display.max_columns', None)\ntaxonomy.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:21:01.270581Z","iopub.execute_input":"2025-03-12T06:21:01.270960Z","iopub.status.idle":"2025-03-12T06:21:01.284724Z","shell.execute_reply.started":"2025-03-12T06:21:01.270922Z","shell.execute_reply":"2025-03-12T06:21:01.283697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ax = taxonomy['class_name'].value_counts()[:20].plot.barh(figsize=(16, 8), color='blue')\nax.set_title('Colombian Animals Class names', size=18, color='red')\nax.set_ylabel('class_name', size=10)\nax.set_xlabel('Count', size=10);","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:21:01.285763Z","iopub.execute_input":"2025-03-12T06:21:01.286178Z","iopub.status.idle":"2025-03-12T06:21:01.550123Z","shell.execute_reply.started":"2025-03-12T06:21:01.286137Z","shell.execute_reply":"2025-03-12T06:21:01.548997Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"taxonomy.groupby(['scientific_name','class_name']).size().reset_index(name='count')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:21:01.551119Z","iopub.execute_input":"2025-03-12T06:21:01.551429Z","iopub.status.idle":"2025-03-12T06:21:01.565702Z","shell.execute_reply.started":"2025-03-12T06:21:01.551402Z","shell.execute_reply":"2025-03-12T06:21:01.564744Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Eunji Goo https://www.kaggle.com/code/quantum09/is-it-going-to-rain\n\nclass_proportion = taxonomy['class_name'].value_counts()/taxonomy['class_name'].value_counts().sum()\ncolormap = plt.cm.tab10(range(0, len(class_proportion)))\nlabels = class_proportion.index\nvalues = class_proportion.values\n\nbars = plt.barh(labels, values)\n\n#plt.xlabel(\"Frequency\") #Não alterou nada\n\n#plt.legend(title='Forest Animals Class names' , bbox_to_anchor=(1.0, 1), loc='lower right')#HORRÌVEL!\n\nbar_plot = class_proportion.plot.barh(color= colormap)\n\n# Add titles, labels, invert y-axis\n\nbar_plot.set_title(\"Colombian Forest Animals by Class names\")\nbar_plot.set_ylabel(\"Class Names\")\n\ntotal = values.sum()\nfor bar, count in zip(bars, values):\n    width = bar.get_width()\n    pct = count / total * 100\n    plt.text(width, bar.get_y() + bar.get_height()/2,\n             f\"{count}\\n({pct:.1f}%)\",\n             ha='left', va='center')\n\n#Invert the axis to have the descending order\nbar_plot.invert_yaxis()\nplt.show(bar_plot)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:21:01.566752Z","iopub.execute_input":"2025-03-12T06:21:01.567115Z","iopub.status.idle":"2025-03-12T06:21:01.768653Z","shell.execute_reply.started":"2025-03-12T06:21:01.567085Z","shell.execute_reply":"2025-03-12T06:21:01.767623Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    import os\n    import librosa\n    import numpy as np\n    import pandas as pd\n    \n    # Set seed\n    np.random.seed(2)\n    \n    # Class labels from train audio\n    class_labels = sorted(os.listdir('/kaggle/input/birdclef-2025/train_audio/'))\n    \n    # List of test soundscapes (only visible during submission)\n    test_soundscape_path = '/kaggle/input/birdclef-2025/test_soundscapes/'\n    test_soundscapes = [os.path.join(test_soundscape_path, afile) for afile in sorted(os.listdir(test_soundscape_path)) if afile.endswith('.ogg')]\n    \n    # Open each soundscape and make predictions for 5-second segments\n    # Use pandas df with 'row_id' plus class labels as columns\n    predictions = pd.DataFrame(columns=['row_id'] + class_labels)\n    for soundscape in test_soundscapes:\n    \n        # Load audio\n        sig, rate = librosa.load(path=soundscape, sr=None)\n    \n        # Split into 5-second chunks\n        chunks = []\n        for i in range(0, len(sig), rate*5):\n            chunk = sig[i:i+rate*5]\n            chunks.append(chunk)\n            \n        # Make predictions for each chunk\n        for i, chunk in enumerate(chunks):\n            \n            # Get row id  (soundscape id + end time of 5s chunk)      \n            row_id = os.path.basename(soundscape).split('.')[0] + f'_{i * 5 + 5}'\n            \n            # Make prediction (let's use random scores for now)\n            # scores = model.predict...\n            scores = np.random.rand(len(class_labels))\n            \n            # Append to predictions as new row\n            new_row = pd.DataFrame([[row_id] + list(scores)], columns=['row_id'] + class_labels)\n            predictions = pd.concat([predictions, new_row], axis=0, ignore_index=True)\n            \n    # Save prediction as csv\n    predictions.to_csv('submission.csv', index=False)\n    predictions.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:21:01.770886Z","iopub.execute_input":"2025-03-12T06:21:01.771220Z","iopub.status.idle":"2025-03-12T06:21:01.824314Z","shell.execute_reply.started":"2025-03-12T06:21:01.771185Z","shell.execute_reply":"2025-03-12T06:21:01.823221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-12T06:21:01.825383Z","iopub.execute_input":"2025-03-12T06:21:01.825685Z","iopub.status.idle":"2025-03-12T06:21:01.831473Z","shell.execute_reply.started":"2025-03-12T06:21:01.825659Z","shell.execute_reply":"2025-03-12T06:21:01.830301Z"}},"outputs":[],"execution_count":null}]}