{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### A few educational resources that help me get started\n- https://www.kaggle.com/code/robikscube/bird-2022-eda-twitch-live-stream\n- https://www.youtube.com/live/MXZKNnuoQXw?si=nId8hCRWbGevTAeW\n- https://towardsdatascience.com/audio-deep-learning-made-simple-sound-classification-step-by-step-cebc936bbe5","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nimport matplotlib.pylab as plt\nimport seaborn as sns\nimport plotly.express as px\n\n# For exploring audio files\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\n\nimport tensorflow as tf\ntf.config.optimizer.set_jit(True) # enable xla for speed up\n\nsns.set_theme(style=\"white\", palette=None)\ncolor_pal = plt.rcParams[\"axes.prop_cycle\"].by_key()[\"color\"]\n\nfrom itertools import cycle\n\ncolor_cycle = cycle(plt.rcParams[\"axes.prop_cycle\"].by_key()[\"color\"])\n\nimport os, sys\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-11T14:42:38.697929Z","iopub.execute_input":"2024-04-11T14:42:38.698283Z","iopub.status.idle":"2024-04-11T14:42:53.286860Z","shell.execute_reply.started":"2024-04-11T14:42:38.698252Z","shell.execute_reply":"2024-04-11T14:42:53.285651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Files\n\nWe are provided with a number of files for this competition. \n\nCSV Files:\n- `train_metadata.csv` - A wide range of metadata is provided for the training data.\n- `sample_submission.csv` - A valid sample submission.\n- `eBird_Taxonomy_v2021.csv` - Data on the relationships between different species.\n\nFolders with Audio Files:\n\n- `train_audio/` - The bulk of the training data consists of short recordings of individual bird calls generously uploaded by users of xenocanto.org.\n- `test_soundscapes/` - When you submit a notebook, the test_soundscapes directory will be populated with approximately 1,100 recordings to be used for scoring.They are 4 minutes long and in ogg audio format.\n- `unlabled_soundscapes/` - Unlabeled audio data from the same recording locations as the test soundscapes.","metadata":{}},{"cell_type":"code","source":"!ls -GFlash --color ../input/birdclef-2024/","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:42:53.289225Z","iopub.execute_input":"2024-04-11T14:42:53.290032Z","iopub.status.idle":"2024-04-11T14:42:54.262667Z","shell.execute_reply.started":"2024-04-11T14:42:53.289995Z","shell.execute_reply":"2024-04-11T14:42:54.261466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read in the CSV files.\nBASE_DIR = '../input/birdclef-2024/'\ntrain = pd.read_csv(f'{BASE_DIR}/train_metadata.csv')\nebird = pd.read_csv(f'{BASE_DIR}/eBird_Taxonomy_v2021.csv')\nss = pd.read_csv(f'{BASE_DIR}/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:42:54.264283Z","iopub.execute_input":"2024-04-11T14:42:54.264678Z","iopub.status.idle":"2024-04-11T14:42:54.539324Z","shell.execute_reply.started":"2024-04-11T14:42:54.264638Z","shell.execute_reply":"2024-04-11T14:42:54.537737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Explore Metadata","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:42:54.541017Z","iopub.execute_input":"2024-04-11T14:42:54.542052Z","iopub.status.idle":"2024-04-11T14:42:54.604680Z","shell.execute_reply.started":"2024-04-11T14:42:54.542013Z","shell.execute_reply":"2024-04-11T14:42:54.603728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.sample(3)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:42:54.607545Z","iopub.execute_input":"2024-04-11T14:42:54.607865Z","iopub.status.idle":"2024-04-11T14:42:54.634859Z","shell.execute_reply.started":"2024-04-11T14:42:54.607838Z","shell.execute_reply":"2024-04-11T14:42:54.633936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ss.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:42:54.636046Z","iopub.execute_input":"2024-04-11T14:42:54.636418Z","iopub.status.idle":"2024-04-11T14:42:54.659765Z","shell.execute_reply.started":"2024-04-11T14:42:54.636385Z","shell.execute_reply":"2024-04-11T14:42:54.658843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Bird counts vary from species to species: some are quite common but others are not due to either low population (e.g., predators) or high difficulty of identification (e.g., intermediate egrets that resemble large and little egrets in appearance)\n\n- Some birds have 500 labels (max?) while others have less than 10","metadata":{}},{"cell_type":"code","source":"fig, axs = plt.subplots(1, 2, figsize=(24, 5))\n# See the frequency of labels in the training dataset\ntrain[\"common_name\"].value_counts().head(25).plot(\n    kind=\"bar\", ax=axs[0], width=.8, color=color_pal[2], rot = 80\n)\n\naxs[0].set_title(\"Top 25 Birds with Labels\", fontsize=20)\n\n# See the frequency of labels in the training dataset\nax = (\n    train[\"common_name\"]\n    .value_counts()\n    .tail(25)\n    .plot(kind=\"bar\", ax=axs[1], width=.8, color=color_pal[1], rot = 80)\n)\naxs[1].set_title(\"Bottom 25 Birds with Labels\", fontsize=20)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:42:54.661054Z","iopub.execute_input":"2024-04-11T14:42:54.661377Z","iopub.status.idle":"2024-04-11T14:42:56.021459Z","shell.execute_reply.started":"2024-04-11T14:42:54.661346Z","shell.execute_reply":"2024-04-11T14:42:56.020497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Where were recordings collected? A map may reveal insights into sampling processes.\n- Training data are collected worldwide: even for the same species, dialects in different geolocaitons may exist -> an issue for classification? accuracy","metadata":{}},{"cell_type":"code","source":"fig = px.scatter_geo(\n    train,\n    lat=\"latitude\",\n    lon=\"longitude\",\n    color=\"common_name\",\n    width=800,\n    height=600,\n    title=\"BirdCLEF 2024 Training Data\",\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:42:56.022570Z","iopub.execute_input":"2024-04-11T14:42:56.022853Z","iopub.status.idle":"2024-04-11T14:42:58.097663Z","shell.execute_reply.started":"2024-04-11T14:42:56.022828Z","shell.execute_reply":"2024-04-11T14:42:58.096768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training Data by Audio Contributors\n\nThere are 1942 different authors in the training dataset. The number of observations per author varies from 1 to 915!\n- 776 of the 1942 authors only have labeled one audio file.\n- each author may use a personalized device, which might accounts for the varivation in ratings","metadata":{}},{"cell_type":"code","source":"author_counts = train[\"author\"].value_counts()\nauthor_counts[author_counts.values == 1]","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:42:58.099018Z","iopub.execute_input":"2024-04-11T14:42:58.099637Z","iopub.status.idle":"2024-04-11T14:42:58.114933Z","shell.execute_reply.started":"2024-04-11T14:42:58.099601Z","shell.execute_reply":"2024-04-11T14:42:58.113483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(1, 1, figsize=(20, 5))\n# See the frequency of labels in the training dataset\ntrain[\"author\"].value_counts().head(50).plot(\n    kind=\"bar\", ax=ax, width=.8, color=color_pal[0], rot = 85\n)\n\nax.set_title(\"Top 50 Contributors\", fontsize=20)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:42:58.116731Z","iopub.execute_input":"2024-04-11T14:42:58.117006Z","iopub.status.idle":"2024-04-11T14:42:59.454914Z","shell.execute_reply.started":"2024-04-11T14:42:58.116983Z","shell.execute_reply":"2024-04-11T14:42:59.453521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Check the quality/rating of soundclips\n- Cornell Orni. Lab collects ratings from eBird users too, so there may be arbituray ratings\n- Most of soundscapes are rated 4 or 5","metadata":{}},{"cell_type":"code","source":"rating_counts = pd.DataFrame(train['rating'].value_counts().sort_index())\nrating_counts.reset_index(inplace = True)\n\nfig = px.bar(\n            rating_counts, x = 'rating', y = 'count',\n            height = 400,\n            width = 800,\n            title = 'rating distribution', \n            text= 'count'\n)\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:42:59.456050Z","iopub.execute_input":"2024-04-11T14:42:59.456335Z","iopub.status.idle":"2024-04-11T14:42:59.550872Z","shell.execute_reply.started":"2024-04-11T14:42:59.456310Z","shell.execute_reply":"2024-04-11T14:42:59.549942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Example Training Audio File","metadata":{}},{"cell_type":"code","source":"# Listen to the audio for the first training example\nfn = train[\"filename\"].values[0]\nprint(fn.split('/')[0])\nipd.Audio(f\"{BASE_DIR}train_audio/{fn}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:42:59.551899Z","iopub.execute_input":"2024-04-11T14:42:59.552168Z","iopub.status.idle":"2024-04-11T14:42:59.575201Z","shell.execute_reply.started":"2024-04-11T14:42:59.552144Z","shell.execute_reply":"2024-04-11T14:42:59.574351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Little Ringed Plover Example - I just spotted a few of them in Kanagawa Prefecture, so lemme compare bird calls (human perception)\n\nfn = train.loc[train[\"common_name\"] == \"Little Ringed Plover\"][\"filename\"].values[0]\nipd.Audio(f\"{BASE_DIR}train_audio/{fn}\")\n\n# no perceivable difference for me: no dialect? maybe no, in light of the migration behavior of these cute golden-eyeringed wading plush","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:42:59.576436Z","iopub.execute_input":"2024-04-11T14:42:59.577056Z","iopub.status.idle":"2024-04-11T14:42:59.608304Z","shell.execute_reply.started":"2024-04-11T14:42:59.577022Z","shell.execute_reply":"2024-04-11T14:42:59.607471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load in the audio file as a numpy array","metadata":{}},{"cell_type":"code","source":"y, sr = librosa.load(f\"{BASE_DIR}train_audio/{fn}\")\nprint(f\"Numpy array of the audio loaded of shape {y.shape} and sample rate {sr}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:42:59.612186Z","iopub.execute_input":"2024-04-11T14:42:59.612571Z","iopub.status.idle":"2024-04-11T14:43:09.078430Z","shell.execute_reply.started":"2024-04-11T14:42:59.612540Z","shell.execute_reply":"2024-04-11T14:43:09.077385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Plot 5 Random Audio Files from the training dataset\n- waveform\n- melspectrum","metadata":{}},{"cell_type":"code","source":"# Plot The Audio File\ndef plot_raw_audio(filename, birdtype, color):\n    y, sr = librosa.load(f\"{BASE_DIR}train_audio/{filename}\")\n    ax = pd.DataFrame(y).plot(\n        figsize=(10, 3), title=f\"{birdtype}\", lw=0.1, color=color\n    )\n    plt.legend().remove()\n    plt.show()\n\n\nfor i, d in train.sample(5, random_state=410).iterrows():\n    plot_raw_audio(d[\"filename\"], d[\"common_name\"], next(color_cycle))","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:43:09.079914Z","iopub.execute_input":"2024-04-11T14:43:09.080695Z","iopub.status.idle":"2024-04-11T14:43:11.743647Z","shell.execute_reply.started":"2024-04-11T14:43:09.080664Z","shell.execute_reply":"2024-04-11T14:43:11.742715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_audio_melspec(filename, birdtype):\n    y, sr = librosa.load(f\"{BASE_DIR}train_audio/{filename}\")\n    S = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=128, fmax=8000)\n\n    fig, ax = plt.subplots(figsize=(10, 3))\n    S_dB = librosa.power_to_db(S, ref=np.max)\n    img = librosa.display.specshow(\n        S_dB, x_axis=\"time\", y_axis=\"mel\", sr=sr, fmax=8000, ax=ax\n    )\n    fig.colorbar(img, ax=ax, format=\"%+2.0f dB\")\n    ax.set(title=f\"Mel-frequency for bird {birdtype}\")\n    plt.show()\n\n\nfor i, d in train.sample(5, random_state=410).iterrows():\n    plot_audio_melspec(d[\"filename\"], d[\"common_name\"])","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:43:11.745019Z","iopub.execute_input":"2024-04-11T14:43:11.745707Z","iopub.status.idle":"2024-04-11T14:43:16.458985Z","shell.execute_reply.started":"2024-04-11T14:43:11.745672Z","shell.execute_reply":"2024-04-11T14:43:16.458117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Peek into test sound data\n- the folder of test sounds is empty unless predictions are submitted\n- so, use unlabled data to perform prediction","metadata":{}},{"cell_type":"code","source":"# Directory containing the test audio files\ntest_audio_dir = '/kaggle/input/birdclef-2024/test_soundscapes/'\n\n# List of file paths for test audio files\ntest_paths = [test_audio_dir + f for f in sorted(os.listdir(test_audio_dir))]\n\n# Checking if there's only one file in the test directory\nif len(test_paths) == 1:\n    # If only one file is found, update the directory to unlabeled_soundscapes\n    test_audio_dir = '/kaggle/input/birdclef-2024/unlabeled_soundscapes/'\n    test_paths = [test_audio_dir + f for f in sorted(os.listdir(test_audio_dir))]\n\n# Creating a DataFrame to store the test file paths and corresponding filenames\ntest_df = pd.DataFrame(test_paths, columns=['filepath'])\n\n# Extracting filenames from file paths and storing them in a new column 'filename'\ntest_df['filename'] = test_df.filepath.map(lambda x: x.split('/')[-1].replace('.ogg', ''))\n\n# Displaying five rows of the DataFrame randomly\ntest_df.sample(5)","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:43:16.460322Z","iopub.execute_input":"2024-04-11T14:43:16.461109Z","iopub.status.idle":"2024-04-11T14:43:16.759949Z","shell.execute_reply.started":"2024-04-11T14:43:16.461074Z","shell.execute_reply":"2024-04-11T14:43:16.759034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:43:16.761166Z","iopub.execute_input":"2024-04-11T14:43:16.761851Z","iopub.status.idle":"2024-04-11T14:43:16.773090Z","shell.execute_reply.started":"2024-04-11T14:43:16.761816Z","shell.execute_reply":"2024-04-11T14:43:16.772086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_audio(filepath, sr=32000, normalize=True):\n    # Load audio file using librosa\n    audio, orig_sr = librosa.load(filepath, sr=None)\n    \n    # Resample audio if specified sample rate differs from original sample rate\n    if sr != orig_sr:\n        audio = librosa.resample(audio, orig_sr, sr)\n    \n    # Convert audio to float32 and flatten the array\n    audio = audio.astype('float32').ravel()\n    \n    # Convert audio to TensorFlow tensor\n    audio = tf.convert_to_tensor(audio)\n    \n    return audio","metadata":{"execution":{"iopub.status.busy":"2024-04-11T14:43:16.774215Z","iopub.execute_input":"2024-04-11T14:43:16.774486Z","iopub.status.idle":"2024-04-11T14:43:16.783333Z","shell.execute_reply.started":"2024-04-11T14:43:16.774462Z","shell.execute_reply":"2024-04-11T14:43:16.782554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_audio(row):\n    # Caption for visualization\n    caption = f'Id: {row.filename}'\n    \n    # Load and preprocess audio\n    audio = load_audio(row.filepath)  # Load audio file\n    audio = audio[:160000]  # Keep fixed length audio = 160,000\n    \n    # Display audio\n    print(\"# Audio:\")\n    display(ipd.Audio(audio.numpy(), rate=32000))  # Display audio using IPython.display\n    print('# Visualization:')\n    \n    # Plot audio wave\n    plt.figure(figsize=(12, 3))\n    plt.title(caption)\n    librosa.display.waveshow(audio.numpy(), sr=32000)  # Plot audio wave using librosa.display\n    plt.xlabel('')\n    plt.title('Waveform')\n    plt.show()\n    \n    # plot melspectrum\n    S = librosa.feature.melspectrogram(y=audio.numpy(), sr=32000, n_mels=128, fmax=8000)\n\n    fig, ax = plt.subplots(figsize=(12, 3))\n    S_dB = librosa.power_to_db(S, ref=np.max)\n    img = librosa.display.specshow(\n        S_dB, x_axis=\"time\", y_axis=\"mel\", sr=sr, fmax=8000, ax=ax\n    )\n    fig.colorbar(img, ax=ax, format=\"%+2.0f dB\")\n    plt.title('Mel Spectrogram')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-11T15:12:25.018859Z","iopub.execute_input":"2024-04-11T15:12:25.019621Z","iopub.status.idle":"2024-04-11T15:12:25.028754Z","shell.execute_reply.started":"2024-04-11T15:12:25.019589Z","shell.execute_reply":"2024-04-11T15:12:25.027675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_audio(test_df.iloc[3])","metadata":{"execution":{"iopub.status.busy":"2024-04-11T15:12:37.287269Z","iopub.execute_input":"2024-04-11T15:12:37.287946Z","iopub.status.idle":"2024-04-11T15:12:38.708585Z","shell.execute_reply.started":"2024-04-11T15:12:37.287914Z","shell.execute_reply":"2024-04-11T15:12:38.707564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}