{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30674,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-04T19:02:40.173908Z","iopub.execute_input":"2024-04-04T19:02:40.174256Z","iopub.status.idle":"2024-04-04T19:03:04.356162Z","shell.execute_reply.started":"2024-04-04T19:02:40.174227Z","shell.execute_reply":"2024-04-04T19:03:04.355115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install --upgrade pip\n","metadata":{"execution":{"iopub.status.busy":"2024-04-04T19:06:13.675389Z","iopub.execute_input":"2024-04-04T19:06:13.676640Z","iopub.status.idle":"2024-04-04T19:06:26.956836Z","shell.execute_reply.started":"2024-04-04T19:06:13.676593Z","shell.execute_reply":"2024-04-04T19:06:26.955687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\n\nimport pandas as pd\npd.options.mode.chained_assignment = None # avoids assignment warning\nimport numpy as np\nimport random\nfrom glob import glob\nfrom tqdm import tqdm\ntqdm.pandas()  # enable progress bars in pandas operations\nimport gc\n\nimport librosa\nimport sklearn\nimport json\n\n# Import for visualization\nimport matplotlib as mpl\n#cmap = mpl.cm.get_cmap('coolwarm')\nimport matplotlib.pyplot as plt\nimport librosa.display as lid\nimport IPython.display as ipd\nimport cv2\n\n# Import KaggleDatasets for accessing Kaggle datasets\nfrom kaggle_datasets import KaggleDatasets\n\n# WandB for experiment tracking\nimport wandb\n\nimport torchaudio\nimport plotly.express as px\nfrom IPython.display import Audio\nfrom shapely.geometry import Point\n\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2024-04-04T19:06:55.822858Z","iopub.execute_input":"2024-04-04T19:06:55.823218Z","iopub.status.idle":"2024-04-04T19:07:08.804301Z","shell.execute_reply.started":"2024-04-04T19:06:55.823190Z","shell.execute_reply":"2024-04-04T19:07:08.803530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the train_metadata.csv file\nmetadata_df = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\n\n# Display the first few rows of the metadata dataframe\nmetadata_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T19:07:44.376835Z","iopub.execute_input":"2024-04-04T19:07:44.377910Z","iopub.status.idle":"2024-04-04T19:07:44.618847Z","shell.execute_reply.started":"2024-04-04T19:07:44.377877Z","shell.execute_reply":"2024-04-04T19:07:44.617857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport torchaudio\nimport IPython.display as ipd\nimport plotly.express as px\n\n# Set the path to the train audio files\ntrain_path = '/kaggle/input/birdclef-2024/train_audio/'\n\n# Load the train_metadata.csv file\nmetadata_df = pd.read_csv(os.path.join('/kaggle/input/birdclef-2024', 'train_metadata.csv'))\n\n# Load the audio file and display it\ndata, rate = torchaudio.load(os.path.join(train_path, metadata_df.filename[0]))\nipd.Audio(data[0, :rate*5], rate=rate)\n\n# Plot the data\npx.line(y=data[0, :rate*5], title=metadata_df.common_name[0]).show()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T19:08:57.874560Z","iopub.execute_input":"2024-04-04T19:08:57.874924Z","iopub.status.idle":"2024-04-04T19:09:01.626107Z","shell.execute_reply.started":"2024-04-04T19:08:57.874898Z","shell.execute_reply":"2024-04-04T19:09:01.624744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import plotly.express as px\n\nfig = px.scatter_mapbox(metadata_df, lat='latitude', lon='longitude', color='primary_label', \n                        hover_name='primary_label', hover_data=['latitude', 'longitude'], \n                        title='Geographical Distribution of Bird Species',\n                        zoom=1, height=600)\nfig.update_layout(mapbox_style=\"open-street-map\")\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-04T19:09:43.716392Z","iopub.execute_input":"2024-04-04T19:09:43.716940Z","iopub.status.idle":"2024-04-04T19:09:44.717633Z","shell.execute_reply.started":"2024-04-04T19:09:43.716902Z","shell.execute_reply":"2024-04-04T19:09:44.716639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Load the eBird_Taxonomy.csv file\neBird_Taxonomy_df = pd.read_csv('/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv')\n\n# Display the first few rows of the eBird_Taxonomy_df DataFrame\nprint(eBird_Taxonomy_df.head())","metadata":{"execution":{"iopub.status.busy":"2024-04-04T19:10:34.965995Z","iopub.execute_input":"2024-04-04T19:10:34.967000Z","iopub.status.idle":"2024-04-04T19:10:35.058713Z","shell.execute_reply.started":"2024-04-04T19:10:34.966965Z","shell.execute_reply":"2024-04-04T19:10:35.057690Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Load the train_metadata.csv file\nmetadata_df = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\n\n# Calculate summary statistics\nstats = metadata_df.describe(include='all')\n\n# Display the summary statistics\nprint(stats)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T19:11:34.686754Z","iopub.execute_input":"2024-04-04T19:11:34.687910Z","iopub.status.idle":"2024-04-04T19:11:34.925439Z","shell.execute_reply.started":"2024-04-04T19:11:34.687868Z","shell.execute_reply":"2024-04-04T19:11:34.924355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Load the train_metadata.csv file\nmetadata_df = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\n\n# Check for missing values\nmissing_values = metadata_df.isnull().sum()\n\n# Display the number of missing values for each column\nprint(missing_values)","metadata":{"execution":{"iopub.status.busy":"2024-04-04T19:12:07.649389Z","iopub.execute_input":"2024-04-04T19:12:07.650114Z","iopub.status.idle":"2024-04-04T19:12:07.784643Z","shell.execute_reply.started":"2024-04-04T19:12:07.650065Z","shell.execute_reply":"2024-04-04T19:12:07.783556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# Load the train_metadata.csv file\nmetadata_df = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\n\n# Plot the distribution of bird species\nsns.countplot(x='primary_label', data=metadata_df, order=metadata_df['primary_label'].value_counts().index)\nplt.xticks(rotation=45)\nplt.title('Distribution of Bird Species')\nplt.xlabel('Bird Species')\nplt.ylabel('Count')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T19:12:37.762532Z","iopub.execute_input":"2024-04-04T19:12:37.762941Z","iopub.status.idle":"2024-04-04T19:12:39.593613Z","shell.execute_reply.started":"2024-04-04T19:12:37.762910Z","shell.execute_reply":"2024-04-04T19:12:39.592480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = px.scatter(metadata_df, x='longitude', y='latitude', color='primary_label', \n                 hover_data=['filename'], title='Geographical Distribution of Bird Species') \nfig.update_layout(legend_title_text='Bird Species', margin=dict(l=0, r=0, t=0, b=0)) \nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T19:13:40.792853Z","iopub.execute_input":"2024-04-04T19:13:40.793221Z","iopub.status.idle":"2024-04-04T19:13:41.982699Z","shell.execute_reply.started":"2024-04-04T19:13:40.793195Z","shell.execute_reply":"2024-04-04T19:13:41.981731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12, 6)) \nsns.countplot(x='author', data=metadata_df, order=metadata_df['author'].value_counts().iloc[:10].index) \nplt.xticks(rotation=45) \nplt.title('Top 10 Authors by Number of Recordings') \nplt.xlabel('Author') \nplt.ylabel('Count') \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-04T19:15:38.941487Z","iopub.execute_input":"2024-04-04T19:15:38.941863Z","iopub.status.idle":"2024-04-04T19:15:39.287106Z","shell.execute_reply.started":"2024-04-04T19:15:38.941832Z","shell.execute_reply":"2024-04-04T19:15:39.286136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read the CSV file into a Pandas DataFrame\ndf_sub = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\")\n\n# Display the first 5 rows of the DataFrame\ndisplay(df_sub.head())","metadata":{"execution":{"iopub.status.busy":"2024-04-04T19:16:46.352644Z","iopub.execute_input":"2024-04-04T19:16:46.353358Z","iopub.status.idle":"2024-04-04T19:16:46.393299Z","shell.execute_reply.started":"2024-04-04T19:16:46.353327Z","shell.execute_reply":"2024-04-04T19:16:46.392181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}