{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-05-26T14:08:22.170999Z","iopub.execute_input":"2024-05-26T14:08:22.171978Z","iopub.status.idle":"2024-05-26T14:08:33.859088Z","shell.execute_reply.started":"2024-05-26T14:08:22.171933Z","shell.execute_reply":"2024-05-26T14:08:33.858056Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Import Libraries**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport librosa\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\n\nimport pandas as pd\npd.options.mode.chained_assignment = None # avoids assignment warning\nimport numpy as np\nimport random\nfrom glob import glob\nfrom tqdm import tqdm\ntqdm.pandas()  # enable progress bars in pandas operations\nimport gc\n\nimport librosa\nimport sklearn\nimport json\n\n# Import for visualization\nimport matplotlib as mpl\n#cmap = mpl.cm.get_cmap('coolwarm')\nimport matplotlib.pyplot as plt\nimport librosa.display as lid\nimport IPython.display as ipd\nimport cv2\n\n# Import KaggleDatasets for accessing Kaggle datasets\nfrom kaggle_datasets import KaggleDatasets\n\n# WandB for experiment tracking\nimport wandb\n\nimport torchaudio\nimport plotly.express as px\nfrom IPython.display import Audio\nfrom shapely.geometry import Point\n\nimport plotly.express as px\n\n# First, load list of audio files by parsing the test_soundscape folder.\ntest_audio_dir = '../input/birdclef-2024/test_soundscapes/'\n\nfile_list = [f for f in sorted(os.listdir(test_audio_dir))]\nfile_list = [file.split('.')[0] for file in file_list if file.endswith('.ogg')]\n\nprint('Number of test soundscapes:', len(file_list))","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:12:19.406542Z","iopub.execute_input":"2024-05-26T14:12:19.407298Z","iopub.status.idle":"2024-05-26T14:12:27.096730Z","shell.execute_reply.started":"2024-05-26T14:12:19.407255Z","shell.execute_reply":"2024-05-26T14:12:27.095489Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# This is where we will store our results\npred = {'row_id': []}\ntrain_audio_dir = '../input/birdclef-2024/train_audio/'\nspecies_list = sorted(os.listdir(train_audio_dir))\nfor species_code in species_list:\n    pred[species_code] = []\n\n# Process audio files and make predictions\nfor afile in file_list:\n    \n    # Complete file path\n    path = test_audio_dir + afile + '.ogg'\n    \n    # Open file with librosa and split signal into 5-second chunks\n    sig, rate = librosa.load(path, sr=32000)\n    # ...\n    \n    # Let's assume we have a list of 48 audio chunks (4min / 5s == 48 segments)\n    chunks = [[] for i in range(48)]\n    \n    # Make prediction for each chunk\n    # Each bird gets a random value in our case\n    # since we don't actually have a model\n    for i in range(len(chunks)):        \n        chunk_end_time = (i + 1) * 5\n        \n        # Assign the row_id which we need to do for each chunk\n        row_id = afile + '_' + str(chunk_end_time)\n        pred['row_id'].append(row_id)\n        \n        for bird in species_list:\n            \n            # This is our random prediction score for this bird\n            score = np.random.uniform()     \n            \n            # Put the result into our prediction dict            \n            pred[bird].append(score)","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:12:36.281423Z","iopub.execute_input":"2024-05-26T14:12:36.282069Z","iopub.status.idle":"2024-05-26T14:12:36.291714Z","shell.execute_reply.started":"2024-05-26T14:12:36.282021Z","shell.execute_reply":"2024-05-26T14:12:36.290543Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Make a new data frame and look at some results        \nresults = pd.DataFrame(pred, columns = ['row_id'] + species_list)\n\n# Quick sanity check\nprint(results.head()) \n    \n# Convert our results to csv\nresults.to_csv(\"submission.csv\", index=False)   ","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:12:44.216443Z","iopub.execute_input":"2024-05-26T14:12:44.216819Z","iopub.status.idle":"2024-05-26T14:12:44.235713Z","shell.execute_reply.started":"2024-05-26T14:12:44.216788Z","shell.execute_reply":"2024-05-26T14:12:44.234579Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pip install --upgrade pip","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:12:48.319949Z","iopub.execute_input":"2024-05-26T14:12:48.320345Z","iopub.status.idle":"2024-05-26T14:13:14.911539Z","shell.execute_reply.started":"2024-05-26T14:12:48.320315Z","shell.execute_reply":"2024-05-26T14:13:14.910150Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# **Data Collection and Processing**","metadata":{}},{"cell_type":"code","source":"# Load the train_metadata.csv file\nmetadata_df = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\n\n# Display the first few rows of the metadata dataframe\nmetadata_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:13:22.654715Z","iopub.execute_input":"2024-05-26T14:13:22.655140Z","iopub.status.idle":"2024-05-26T14:13:22.844745Z","shell.execute_reply.started":"2024-05-26T14:13:22.655101Z","shell.execute_reply":"2024-05-26T14:13:22.843818Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_path = '/kaggle/input/birdclef-2024/train_audio/'\ndata, rate = torchaudio.load(train_path + metadata_df.filename[0])\ndisplay(Audio(data[0, :rate*40], rate=rate))\npx.line(y=data[0, :rate*40], title=metadata_df.common_name[0])","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:13:28.493415Z","iopub.execute_input":"2024-05-26T14:13:28.493799Z","iopub.status.idle":"2024-05-26T14:13:35.210271Z","shell.execute_reply.started":"2024-05-26T14:13:28.493767Z","shell.execute_reply":"2024-05-26T14:13:35.208666Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Assuming metadata_df contains the data\n# Create scatter plot on a map\n\nfig = px.scatter_mapbox(metadata_df, lat='latitude', lon='longitude', color='primary_label', \n                        hover_name='primary_label', hover_data=['latitude', 'longitude'], \n                        title='Geographical Distribution of Bird Species',\n                        zoom=1, height=800)\nfig.update_layout(mapbox_style=\"open-street-map\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:16:20.178913Z","iopub.execute_input":"2024-05-26T14:16:20.179380Z","iopub.status.idle":"2024-05-26T14:16:21.007005Z","shell.execute_reply.started":"2024-05-26T14:16:20.179347Z","shell.execute_reply":"2024-05-26T14:16:21.005856Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load the train_metadata.csv file\neBird_Taxonomy_df = pd.read_csv('/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv')\n\n# Display the first few rows of the metadata dataframe\neBird_Taxonomy_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:16:26.212480Z","iopub.execute_input":"2024-05-26T14:16:26.212858Z","iopub.status.idle":"2024-05-26T14:16:26.297375Z","shell.execute_reply.started":"2024-05-26T14:16:26.212829Z","shell.execute_reply":"2024-05-26T14:16:26.296228Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Summary statistics\nmetadata_df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:16:36.296977Z","iopub.execute_input":"2024-05-26T14:16:36.297363Z","iopub.status.idle":"2024-05-26T14:16:36.325184Z","shell.execute_reply.started":"2024-05-26T14:16:36.297334Z","shell.execute_reply":"2024-05-26T14:16:36.324103Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"metadata_df\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:16:41.175430Z","iopub.execute_input":"2024-05-26T14:16:41.175783Z","iopub.status.idle":"2024-05-26T14:16:41.196051Z","shell.execute_reply.started":"2024-05-26T14:16:41.175756Z","shell.execute_reply":"2024-05-26T14:16:41.195091Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import plotly.graph_objects as go\n\nfig = go.Figure(go.Scattermapbox(\n    mode = \"markers+lines\",\n    lon = [10, 20, 30],\n    lat = [10, 20,30],\n    marker = {'size': 10}))\n\nfig.add_trace(go.Scattermapbox(\n    mode = \"markers+lines\",\n    lon = [-50, -60,40],\n    lat = [30, 10, -20],\n    marker = {'size': 10}))\n\nfig.update_layout(\n    margin ={'l':0,'t':0,'b':0,'r':0},\n    mapbox = {\n        'center': {'lon': 10, 'lat': 10},\n        'style': \"open-street-map\",\n        'center': {'lon': -20, 'lat': -20},\n        'zoom': 1})\n\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:16:48.546851Z","iopub.execute_input":"2024-05-26T14:16:48.547260Z","iopub.status.idle":"2024-05-26T14:16:48.574098Z","shell.execute_reply.started":"2024-05-26T14:16:48.547226Z","shell.execute_reply":"2024-05-26T14:16:48.572945Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for missing values\nmetadata_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:16:52.388598Z","iopub.execute_input":"2024-05-26T14:16:52.388963Z","iopub.status.idle":"2024-05-26T14:16:52.408846Z","shell.execute_reply.started":"2024-05-26T14:16:52.388934Z","shell.execute_reply":"2024-05-26T14:16:52.407715Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distribution of bird species in the training data\nplt.figure(figsize=(12, 6))\nsns.countplot(x='primary_label', data=metadata_df, order=metadata_df['primary_label'].value_counts().index)\nplt.xticks(rotation=45)\nplt.title('Distribution of Bird Species')\nplt.xlabel('Bird Species')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-14T17:37:44.757518Z","iopub.execute_input":"2024-05-14T17:37:44.758042Z","iopub.status.idle":"2024-05-14T17:37:46.670887Z","shell.execute_reply.started":"2024-05-14T17:37:44.758000Z","shell.execute_reply":"2024-05-14T17:37:46.669647Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Relationship between latitude and longitude\nimport folium as fl\nfrom folium.plugins import HeatMap\nplt.figure(figsize=(10, 8))\nsns.scatterplot(x='longitude', y='latitude', data=metadata_df, hue='primary_label', palette='viridis', alpha=0.5)\nplt.title('Geographical Distribution of Bird Species')\nplt.xlabel('Longitude')\nplt.ylabel('Latitude')\nplt.legend(title='Bird Species', loc='upper right', bbox_to_anchor=(1.25, 1))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:16:57.081743Z","iopub.execute_input":"2024-05-26T14:16:57.082152Z","iopub.status.idle":"2024-05-26T14:17:02.896006Z","shell.execute_reply.started":"2024-05-26T14:16:57.082118Z","shell.execute_reply":"2024-05-26T14:17:02.895015Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Distribution of recordings by authors\nplt.figure(figsize=(12, 6))\nsns.countplot(x='author', data=metadata_df, order=metadata_df['author'].value_counts().index[:10])\nplt.xticks(rotation=45)\nplt.title('Top 10 Authors by Number of Recordings')\nplt.xlabel('Author')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:21:25.519984Z","iopub.execute_input":"2024-05-26T14:21:25.520722Z","iopub.status.idle":"2024-05-26T14:21:25.834051Z","shell.execute_reply.started":"2024-05-26T14:21:25.520689Z","shell.execute_reply":"2024-05-26T14:21:25.832936Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## **Submission**\nWe retrain the model on the full training data and create a submission file.","metadata":{}},{"cell_type":"code","source":"df_sub = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\")\ndisplay(df_sub)","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:21:30.069583Z","iopub.execute_input":"2024-05-26T14:21:30.070265Z","iopub.status.idle":"2024-05-26T14:21:30.103462Z","shell.execute_reply.started":"2024-05-26T14:21:30.070231Z","shell.execute_reply":"2024-05-26T14:21:30.102416Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-26T14:21:37.687422Z","iopub.execute_input":"2024-05-26T14:21:37.687798Z","iopub.status.idle":"2024-05-26T14:21:37.695459Z","shell.execute_reply.started":"2024-05-26T14:21:37.687768Z","shell.execute_reply":"2024-05-26T14:21:37.694428Z"},"trusted":true},"outputs":[],"execution_count":null}]}