{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-08-01T15:45:43.264390Z","iopub.execute_input":"2024-08-01T15:45:43.264762Z","iopub.status.idle":"2024-08-01T15:46:02.082993Z","shell.execute_reply.started":"2024-08-01T15:45:43.264726Z","shell.execute_reply":"2024-08-01T15:46:02.081947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Import Libraries**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport librosa\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\n\nimport pandas as pd\npd.options.mode.chained_assignment = None # avoids assignment warning\nimport numpy as np\nimport random\nfrom glob import glob\nfrom tqdm import tqdm\ntqdm.pandas()  # enable progress bars in pandas operations\nimport gc\n\nimport librosa\nimport sklearn\nimport json\n\n# Import for visualization\nimport matplotlib as mpl\n#cmap = mpl.cm.get_cmap('coolwarm')\nimport matplotlib.pyplot as plt\nimport librosa.display as lid\nimport IPython.display as ipd\nimport cv2\n\n# Import KaggleDatasets for accessing Kaggle datasets\nfrom kaggle_datasets import KaggleDatasets\n\n# WandB for experiment tracking\nimport wandb\n\nimport torchaudio\nimport plotly.express as px\nfrom IPython.display import Audio\nfrom shapely.geometry import Point\n\nimport plotly.express as px\n\n# First, load list of audio files by parsing the test_soundscape folder.\ntest_audio_dir = '../input/birdclef-2024/test_soundscapes/'\n\nfile_list = [f for f in sorted(os.listdir(test_audio_dir))]\nfile_list = [file.split('.')[0] for file in file_list if file.endswith('.ogg')]\n\nprint('Number of test soundscapes:', len(file_list))","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:02.085023Z","iopub.execute_input":"2024-08-01T15:46:02.085580Z","iopub.status.idle":"2024-08-01T15:46:09.821227Z","shell.execute_reply.started":"2024-08-01T15:46:02.085542Z","shell.execute_reply":"2024-08-01T15:46:09.820168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This is where we will store our results\npred = {'row_id': []}\ntrain_audio_dir = '../input/birdclef-2024/train_audio/'\nspecies_list = sorted(os.listdir(train_audio_dir))\nfor species_code in species_list:\n    pred[species_code] = []\n\n# Process audio files and make predictions\nfor afile in file_list:\n    \n    # Complete file path\n    path = test_audio_dir + afile + '.ogg'\n    \n    # Open file with librosa and split signal into 5-second chunks\n    sig, rate = librosa.load(path, sr=32000)\n    # ...\n    \n    # Let's assume we have a list of 48 audio chunks (4min / 5s == 48 segments)\n    chunks = [[] for i in range(48)]\n    \n    # Make prediction for each chunk\n    # Each bird gets a random value in our case\n    # since we don't actually have a model\n    for i in range(len(chunks)):        \n        chunk_end_time = (i + 1) * 5\n        \n        # Assign the row_id which we need to do for each chunk\n        row_id = afile + '_' + str(chunk_end_time)\n        pred['row_id'].append(row_id)\n        \n        for bird in species_list:\n            \n            # This is our random prediction score for this bird\n            score = np.random.uniform()     \n            \n            # Put the result into our prediction dict            \n            pred[bird].append(score)","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:09.822502Z","iopub.execute_input":"2024-08-01T15:46:09.823052Z","iopub.status.idle":"2024-08-01T15:46:09.832179Z","shell.execute_reply.started":"2024-08-01T15:46:09.823021Z","shell.execute_reply":"2024-08-01T15:46:09.831112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make a new data frame and look at some results        \nresults = pd.DataFrame(pred, columns = ['row_id'] + species_list)\n\n# Quick sanity check\nprint(results.head()) \n    \n# Convert our results to csv\nresults.to_csv(\"submission.csv\", index=False)   ","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:09.834332Z","iopub.execute_input":"2024-08-01T15:46:09.834725Z","iopub.status.idle":"2024-08-01T15:46:09.862104Z","shell.execute_reply.started":"2024-08-01T15:46:09.834698Z","shell.execute_reply":"2024-08-01T15:46:09.860918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install --upgrade pip","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:09.863631Z","iopub.execute_input":"2024-08-01T15:46:09.863959Z","iopub.status.idle":"2024-08-01T15:46:35.858304Z","shell.execute_reply.started":"2024-08-01T15:46:09.863933Z","shell.execute_reply":"2024-08-01T15:46:35.857031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data Collection and Processing**","metadata":{}},{"cell_type":"code","source":"# Load the train_metadata.csv file\nmetadata_df = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\n\n# Display the first few rows of the metadata dataframe\nmetadata_df.head(1)","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:35.859975Z","iopub.execute_input":"2024-08-01T15:46:35.860356Z","iopub.status.idle":"2024-08-01T15:46:36.049614Z","shell.execute_reply.started":"2024-08-01T15:46:35.860322Z","shell.execute_reply":"2024-08-01T15:46:36.048605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '/kaggle/input/birdclef-2024/train_audio/'\ndata, rate = torchaudio.load(train_path + metadata_df.filename[0])\ndisplay(Audio(data[0, :rate*40], rate=rate))\npx.line(y=data[0, :rate*40], title=metadata_df.common_name[0])","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:36.050834Z","iopub.execute_input":"2024-08-01T15:46:36.051144Z","iopub.status.idle":"2024-08-01T15:46:42.837482Z","shell.execute_reply.started":"2024-08-01T15:46:36.051119Z","shell.execute_reply":"2024-08-01T15:46:42.835798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Assuming metadata_df contains the data\n# Create scatter plot on a map\n\nfig = px.scatter_mapbox(metadata_df, lat='latitude', lon='longitude', color='primary_label', \n                        hover_name='primary_label', hover_data=['latitude', 'longitude'], \n                        title='Geographical Distribution of Bird Species',\n                        zoom=1, height=800)\nfig.update_layout(mapbox_style=\"open-street-map\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:42.838969Z","iopub.execute_input":"2024-08-01T15:46:42.839345Z","iopub.status.idle":"2024-08-01T15:46:43.717521Z","shell.execute_reply.started":"2024-08-01T15:46:42.839315Z","shell.execute_reply":"2024-08-01T15:46:43.716510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the train_metadata.csv file\neBird_Taxonomy_df = pd.read_csv('/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv')\n\n# Display the first few rows of the metadata dataframe\neBird_Taxonomy_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:43.718903Z","iopub.execute_input":"2024-08-01T15:46:43.719283Z","iopub.status.idle":"2024-08-01T15:46:43.810107Z","shell.execute_reply.started":"2024-08-01T15:46:43.719251Z","shell.execute_reply":"2024-08-01T15:46:43.809122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Summary statistics\nmetadata_df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:43.813400Z","iopub.execute_input":"2024-08-01T15:46:43.813724Z","iopub.status.idle":"2024-08-01T15:46:43.838220Z","shell.execute_reply.started":"2024-08-01T15:46:43.813698Z","shell.execute_reply":"2024-08-01T15:46:43.837010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df\n\n\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:43.839598Z","iopub.execute_input":"2024-08-01T15:46:43.839920Z","iopub.status.idle":"2024-08-01T15:46:43.861611Z","shell.execute_reply.started":"2024-08-01T15:46:43.839894Z","shell.execute_reply":"2024-08-01T15:46:43.860394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.graph_objects as go\n\nfig = go.Figure(go.Scattermapbox(\n    mode = \"markers+lines\",\n    lon = [10, 20, 30],\n    lat = [10, 20,30],\n    marker = {'size': 10}))\n\nfig.add_trace(go.Scattermapbox(\n    mode = \"markers+lines\",\n    lon = [-50, -60,40],\n    lat = [30, 10, -20],\n    marker = {'size': 10}))\n\nfig.update_layout(\n    margin ={'l':0,'t':0,'b':0,'r':0},\n    mapbox = {\n        'center': {'lon': 10, 'lat': 10},\n        'style': \"open-street-map\",\n        'center': {'lon': -20, 'lat': -20},\n        'zoom': 1})\n\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:43.863112Z","iopub.execute_input":"2024-08-01T15:46:43.863538Z","iopub.status.idle":"2024-08-01T15:46:43.889192Z","shell.execute_reply.started":"2024-08-01T15:46:43.863501Z","shell.execute_reply":"2024-08-01T15:46:43.888294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for missing values\nmetadata_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:43.890622Z","iopub.execute_input":"2024-08-01T15:46:43.891121Z","iopub.status.idle":"2024-08-01T15:46:43.910712Z","shell.execute_reply.started":"2024-08-01T15:46:43.891086Z","shell.execute_reply":"2024-08-01T15:46:43.909610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution of bird species in the training data\nplt.figure(figsize=(12, 6))\nsns.countplot(x='primary_label', data=metadata_df, order=metadata_df['primary_label'].value_counts().index)\nplt.xticks(rotation=45)\nplt.title('Distribution of Bird Species')\nplt.xlabel('Bird Species')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:43.911902Z","iopub.execute_input":"2024-08-01T15:46:43.912213Z","iopub.status.idle":"2024-08-01T15:46:45.475420Z","shell.execute_reply.started":"2024-08-01T15:46:43.912188Z","shell.execute_reply":"2024-08-01T15:46:45.474446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Relationship between latitude and longitude\nimport folium as fl\nfrom folium.plugins import HeatMap\nplt.figure(figsize=(10, 8))\nsns.scatterplot(x='longitude', y='latitude', data=metadata_df, hue='primary_label', palette='viridis', alpha=0.5)\nplt.title('Geographical Distribution of Bird Species')\nplt.xlabel('Longitude')\nplt.ylabel('Latitude')\nplt.legend(title='Bird Species', loc='upper right', bbox_to_anchor=(1.25, 1))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:45.476696Z","iopub.execute_input":"2024-08-01T15:46:45.477068Z","iopub.status.idle":"2024-08-01T15:46:51.184570Z","shell.execute_reply.started":"2024-08-01T15:46:45.477026Z","shell.execute_reply":"2024-08-01T15:46:51.183441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution of recordings by authors\nplt.figure(figsize=(12, 6))\nsns.countplot(x='author', data=metadata_df, order=metadata_df['author'].value_counts().index[:10])\nplt.xticks(rotation=45)\nplt.title('Top 10 Authors by Number of Recordings')\nplt.xlabel('Author')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:51.185858Z","iopub.execute_input":"2024-08-01T15:46:51.186557Z","iopub.status.idle":"2024-08-01T15:46:51.497067Z","shell.execute_reply.started":"2024-08-01T15:46:51.186520Z","shell.execute_reply":"2024-08-01T15:46:51.496087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Submission**\nWe retrain the model on the full training data and create a submission file.","metadata":{}},{"cell_type":"code","source":"df_sub = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\")\ndisplay(df_sub)","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:46:51.498313Z","iopub.execute_input":"2024-08-01T15:46:51.498630Z","iopub.status.idle":"2024-08-01T15:46:51.533263Z","shell.execute_reply.started":"2024-08-01T15:46:51.498605Z","shell.execute_reply":"2024-08-01T15:46:51.532282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-08-01T15:52:10.139392Z","iopub.execute_input":"2024-08-01T15:52:10.139843Z","iopub.status.idle":"2024-08-01T15:52:10.148821Z","shell.execute_reply.started":"2024-08-01T15:52:10.139812Z","shell.execute_reply":"2024-08-01T15:52:10.147777Z"},"trusted":true},"execution_count":null,"outputs":[]}]}