{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n# Birdclef 2024 - Audio Classification\n\n\n<center><img src=\"../images/cover_01.jpg\"/></center>\n\n## Introduction\n\nThis is a simple notebook to get started with the Birdclef 2024 competition. The goal of this competition is to classify bird sounds into different categories.\n\n\n## Import Libraries\n","metadata":{}},{"cell_type":"code","source":"# Standard data manipulation libraries\nimport numpy as np\nimport pandas as pd\nimport os\nimport pprint\nimport random\nimport joblib # Used for exporting and importing trained models\nimport librosa # Used for processing audio\n\n# Visualization libraries\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nfrom IPython.display import Audio # Used for displaying an interactive audio player\n\nKAGGLE = True","metadata":{"ExecuteTime":{"end_time":"2024-06-08T14:15:09.338723Z","start_time":"2024-06-08T14:15:09.326079Z"},"execution":{"iopub.status.busy":"2024-06-10T15:16:32.564649Z","iopub.execute_input":"2024-06-10T15:16:32.565064Z","iopub.status.idle":"2024-06-10T15:16:34.666464Z","shell.execute_reply.started":"2024-06-10T15:16:32.565030Z","shell.execute_reply":"2024-06-10T15:16:34.665292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import data\n\nIf kaggle is used, the data is already available in the input folder. If not, the data can be downloaded from the competition page.","metadata":{}},{"cell_type":"code","source":"if KAGGLE: \n    DATA_DIR = '/kaggle/input/birdclef-2024/'\nelse:\n    DATA_DIR = '../../kaggle/input/birdclef-2024/'\n    \ndf_metadata = pd.read_csv(DATA_DIR + 'train_metadata.csv')\ndf_taxonomy = pd.read_csv(DATA_DIR + 'eBird_Taxonomy_v2021.csv')\nsample_submission = pd.read_csv(DATA_DIR + 'sample_submission.csv')","metadata":{"ExecuteTime":{"end_time":"2024-06-08T14:15:09.434381Z","start_time":"2024-06-08T14:15:09.341146Z"},"execution":{"iopub.status.busy":"2024-06-10T15:16:34.668867Z","iopub.execute_input":"2024-06-10T15:16:34.669481Z","iopub.status.idle":"2024-06-10T15:16:34.972618Z","shell.execute_reply.started":"2024-06-10T15:16:34.669446Z","shell.execute_reply":"2024-06-10T15:16:34.971341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find the number of audio files in the train_audio folder\naudio_files = []\nfor root, dirs, files in os.walk(DATA_DIR + 'train_audio'):\n    audio_files.extend(files)\nprint(f\"Number of audio files: {len(audio_files)}\")\n\n# Find the number of unique filenames in the metadata\nunique_filenames = df_metadata['filename'].nunique()\nprint(f\"Number of unique filenames: {unique_filenames}\")\n\nassert len(audio_files) == unique_filenames, \"Number of audio files and unique filenames do not match\"\n\n# # Show a couple of rows of the data\ndisplay(df_metadata.head(2))\ndisplay(df_taxonomy.head(2))\n\n","metadata":{"ExecuteTime":{"end_time":"2024-06-08T14:15:09.527813Z","start_time":"2024-06-08T14:15:09.434954Z"},"execution":{"iopub.status.busy":"2024-06-10T15:16:34.974534Z","iopub.execute_input":"2024-06-10T15:16:34.975035Z","iopub.status.idle":"2024-06-10T15:16:40.042278Z","shell.execute_reply.started":"2024-06-10T15:16:34.974992Z","shell.execute_reply":"2024-06-10T15:16:40.040943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Exploration\n\nAn audio file is a sequence of samples that represent the sound. The sampling \nrate is the number of samples per second. The sampling rate is usually 44.1 kHz\nfor music, which means that 44,100 samples are taken per second. The duration of\nthe audio file is the number of samples divided by the sampling rate.\n\nIdeally, We could use sequence classification models like RNNs or CNNs to \nclassify audio data. However, thanks to a blog from [Google Research](https://research.google/blog/separating-birdsong-in-the-wild-for-classification/),\nwe can use a simpler approach. The blog suggests that we can use a spectrogram \nof the audio data to classify bird sounds. A spectrogram is a visual \nrepresentation of the spectrum of frequencies of a signal as it varies with \ntime.\n\n","metadata":{}},{"cell_type":"code","source":"class AudioAnalyzer:\n    def __init__(self, DATA_DIR, df_metadata, audio_file=None):\n        self.DATA_DIR = DATA_DIR\n        self.df_metadata = df_metadata\n        if audio_file is None:\n            # Get a random audio file\n            self.audio_file, self.audio_metadata = self.get_random_audio_file()\n        else:\n            # Use the provided audio file\n            self.audio_file = audio_file\n            primary_label, filename = os.path.split(audio_file)\n            primary_label = os.path.basename(primary_label)\n            # Find the row in the DataFrame that corresponds to this audio file\n            self.audio_metadata = df_metadata[(df_metadata['filename'] == f\"{primary_label}/{filename}\")].iloc[0]\n        self.audio_data, self.sampling_rate = librosa.load(self.audio_file, sr=None)\n        self.plot_figure_size = (30, 10)\n        self.plot_title_size = 20\n\n    def get_random_audio_file(self):\n        # Select a random row from the DataFrame\n        row = self.df_metadata.sample(1).iloc[0]\n        # Construct the file path from the primary_label and filename fields\n        audio_file = os.path.join(self.DATA_DIR, 'train_audio', row['filename'])\n        return audio_file, row\n    \n    def show_metadata(self):\n        # If the metadata was found, return its data\n        if self.audio_metadata is not None:\n            return self.audio_metadata.to_dict()\n            # return pd.DataFrame([self.audio_metadata])\n        else:\n            print(\"Metadata for this audio file was not found.\")\n\n    def play_audio(self):\n        return Audio(self.audio_file)\n\n    def show_waveform(self):\n        plt.figure(figsize=self.plot_figure_size)\n        plt.plot(self.audio_data)\n        plt.title('Waveform', fontsize=self.plot_title_size)\n        plt.xlabel('Time')\n        plt.ylabel('Amplitude')\n\n    def show_spectrogram(self):\n        spectrogram = librosa.stft(self.audio_data)\n        spectrogram_db = librosa.amplitude_to_db(abs(spectrogram))\n        plt.figure(figsize=self.plot_figure_size)\n        librosa.display.specshow(spectrogram_db, sr=self.sampling_rate, x_axis='time', y_axis='log')\n        plt.colorbar(format='%+2.0f dB')\n        plt.title('Spectrogram (dB)', fontsize=self.plot_title_size)\n\n    def show_mel_spectrogram(self):\n        mel_spectrogram = librosa.feature.melspectrogram(y=self.audio_data, sr=self.sampling_rate)\n        plt.figure(figsize=self.plot_figure_size)\n        librosa.display.specshow(librosa.power_to_db(mel_spectrogram, ref=np.max), x_axis='time', y_axis='mel', sr=self.sampling_rate)\n        plt.colorbar(format='%+2.0f dB')\n        plt.title('Mel spectrogram', fontsize=self.plot_title_size)\n\n    def show_chromagram(self):\n        chromagram = librosa.feature.chroma_stft(y=self.audio_data, sr=self.sampling_rate)\n        plt.figure(figsize=self.plot_figure_size)\n        librosa.display.specshow(chromagram, x_axis='time', y_axis='chroma', sr=self.sampling_rate, cmap='coolwarm')\n        plt.colorbar()\n        plt.title('Chromagram', fontsize=self.plot_title_size)\n\n    def show_mfcc(self):\n        mfcc = librosa.feature.mfcc(y=self.audio_data, sr=self.sampling_rate, n_mfcc=13)\n        plt.figure(figsize=self.plot_figure_size)\n        librosa.display.specshow(mfcc, x_axis='time', sr=self.sampling_rate)\n        plt.colorbar()\n        plt.title('MFCC', fontsize=self.plot_title_size)\n\n    def show_all(self):\n        self.show_waveform()\n        self.show_spectrogram()\n        self.show_mel_spectrogram()\n        self.show_chromagram()\n        self.show_mfcc()\n        plt.tight_layout()\n        plt.show()\n\n\n# file_name = os.path.join(DATA_DIR, 'train_audio', 'ashdro1', 'XC492732.ogg')\n# analyzer = AudioAnalyzer(DATA_DIR, df_metadata, file_name)\n\nanalyzer = AudioAnalyzer(DATA_DIR, df_metadata)\npprint.pprint(analyzer.show_metadata())\ndisplay(analyzer.play_audio())\nanalyzer.show_all()","metadata":{"ExecuteTime":{"end_time":"2024-06-08T14:15:11.328146Z","start_time":"2024-06-08T14:15:09.529082Z"},"execution":{"iopub.status.busy":"2024-06-10T15:16:40.044047Z","iopub.execute_input":"2024-06-10T15:16:40.044488Z","iopub.status.idle":"2024-06-10T15:17:05.064198Z","shell.execute_reply.started":"2024-06-10T15:16:40.044456Z","shell.execute_reply":"2024-06-10T15:17:05.062726Z"},"trusted":true},"execution_count":null,"outputs":[]}]}