{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Change Log\n\n**Version 0:** Add metadata EDA, audio samples EDA, image features (STFT, Mel Spectrograms, MFCC)<br>\n**Version 1:** Add samples outliers analysis based on audio duration, made some refactoring<br>\n**Version 2:** Add noise filtering investigations, created AudioProcessor class<br>","metadata":{}},{"cell_type":"code","source":"import os\nimport glob\nimport random\nimport pywt\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nimport plotly\nimport plotly.express as px\nimport plotly.graph_objs as go\nfrom plotly.offline import init_notebook_mode\n\nimport librosa\n\nfrom tqdm import tqdm\nfrom scipy.signal import butter, filtfilt\nfrom IPython.display import Audio, display\n\npd.set_option(\"display.max_rows\", 999)\npd.set_option(\"display.max_columns\", 999)\nplt.style.use(\"ggplot\")\n\ninit_notebook_mode(connected=True)\n\nseed = 42\nrandom.seed(seed)\nos.environ[\"PYTHONHASHSEED\"] = str(seed)\nnp.random.seed(seed)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-23T12:08:18.510775Z","iopub.execute_input":"2024-04-23T12:08:18.511211Z","iopub.status.idle":"2024-04-23T12:08:18.524137Z","shell.execute_reply.started":"2024-04-23T12:08:18.511183Z","shell.execute_reply":"2024-04-23T12:08:18.522227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helpers","metadata":{}},{"cell_type":"code","source":"def load_audio(audio_path, **kwargs):\n    sr = kwargs.get(\"sr\", 32000)\n    mono = kwargs.get(\"mono\", True)\n    offset = kwargs.get(\"offset\", 0.0)\n    duration = kwargs.get(\"duration\", None)\n    y, sr = librosa.load(audio_path, sr=sr, mono=mono, offset=offset, duration=duration)\n    return y, sr\n\ndef display_audio(y, sr, title=None):\n    display(Audio(y, rate=sr))\n\n    plt.figure(figsize=(10, 4))\n    librosa.display.waveshow(y, sr=sr)\n    if title:\n        plt.title(title)\n    else:\n        plt.title(f'sr={sr}')\n    plt.xlabel('Time (s)')\n    plt.ylabel('Amplitude')\n    plt.show()\n\ndef wavelet_filter(y, thresh=0.1, wavelet=\"db4\", level=None):\n    thresh = thresh * np.nanmax(y)\n    coeffs = pywt.wavedec(y, wavelet, level=level, mode=\"per\")\n    coeffs = [pywt.threshold(c, value=thresh, mode=\"soft\") for c in coeffs]\n    reconstructed_signal = pywt.waverec(coeffs, wavelet, mode=\"per\")\n    return reconstructed_signal\n\ndef butter_bandpass_filter(y, lowcut, highcut, sr, order=5):\n    def butter_bandpass(lowcut, highcut, sr, order=5):\n        nyq = 0.5 * sr # Nyquist frequency calculation\n        low, high = lowcut / nyq, highcut / nyq\n        b, a = butter(order, [low, high], btype='band')\n        return b, a\n\n    b, a = butter_bandpass(lowcut, highcut, sr, order=order)\n    y = filtfilt(b, a, y)\n    return y\n    \ndef calculate_stft(y, n_fft, hop_length):\n    S = librosa.stft(y, n_fft=n_fft, hop_length=hop_length)\n    Sxx, P = np.abs(S), np.angle(S)\n    Sxx_db = librosa.amplitude_to_db(Sxx, ref=np.max)\n    return Sxx_db, P\n\ndef calculate_mel_spectrogram(y, sr, n_fft, hop_length, n_mels):\n    S_mel = librosa.feature.melspectrogram(y=y, sr=sr, n_fft=n_fft, hop_length=hop_length, n_mels=n_mels)\n    S_mel_db = librosa.power_to_db(S_mel, ref=np.max)\n    return S_mel_db\n\ndef calculate_mfcc(y, sr, n_mfcc):\n    # MFCC (Mel-Frequency Cepstral Coefficients)\n    S_mfcc = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=n_mfcc)\n    return S_mfcc\n\ndef plot_spectrograms(y, sr, n_fft, hop_length, n_mels, n_mfcc):\n    Sxx_db, P = calculate_stft(y, n_fft, hop_length)\n    S_mel_db = calculate_mel_spectrogram(y, sr, n_fft, hop_length, n_mels)\n    S_mfcc = calculate_mfcc(y, sr, n_mfcc)\n    \n    plt.figure(figsize=(14, 10))\n\n    plt.subplot(2, 2, 1)\n    stft_mag_db = librosa.display.specshow(Sxx_db, sr=sr, x_axis='time', y_axis='log')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title('Magnitude STFT Spectrogram')\n\n    plt.subplot(2, 2, 2)\n    stft_phase = librosa.display.specshow(P, sr=sr, x_axis='time', y_axis='log', cmap='hsv')\n    plt.colorbar(format='%+2.0f rad')\n    plt.title('Phase STFT Spectrogram')\n\n    plt.subplot(2, 2, 3)\n    s_mel_db = librosa.display.specshow(S_mel_db, sr=sr, hop_length=hop_length, x_axis='time', y_axis='mel')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title('Mel Spectrogram')\n\n    plt.subplot(2, 2, 4)\n    s_mfcc = librosa.display.specshow(S_mfcc, sr=sr, x_axis='time')\n    plt.colorbar()\n    plt.title('MFCC')\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:08:18.533520Z","iopub.execute_input":"2024-04-23T12:08:18.533904Z","iopub.status.idle":"2024-04-23T12:08:18.558843Z","shell.execute_reply.started":"2024-04-23T12:08:18.533875Z","shell.execute_reply":"2024-04-23T12:08:18.556896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AudioProcessor:\n    def __init__(self, audio_path, sample_rate=32000, **kwargs):\n        self.audio_path = audio_path\n        self.sr = sample_rate\n\n        self.mono = kwargs.get(\"mono\", True)\n        self.offset = kwargs.get(\"offset\", 0.0)\n        self.duration = kwargs.get(\"duration\", None)\n\n        self.normalize_signal = kwargs.get(\"normalize_signal\", False)\n        self.filter_kind = kwargs.get(\"filter_kind\", \"\")\n        self.n_fft = kwargs.get(\"n_fft\", 512)\n        self.hop_length = kwargs.get(\"hop_length\", 256)\n        self.n_mels = kwargs.get(\"n_mels\", 64)\n        self.n_mfcc = kwargs.get(\"n_mfcc\", 64)\n        self.save_stft = kwargs.get(\"save_stft\", False)\n        self.save_mel = kwargs.get(\"save_mel\", False)\n        self.save_mfcc = kwargs.get(\"save_mfcc\", False)\n        self.save_dir = kwargs.get(\"save_dir\", f\"./train/{audio_path.rsplit('/')[-2]}\")\n        os.makedirs(self.save_dir, exist_ok=True)\n        self.audio_name = kwargs.get(\"audio_name\", os.path.splitext(os.path.basename(audio_path))[0])\n        \n        self.y, self.sr = self.load_audio()\n        \n    def load_audio(self):\n        y, sr = librosa.load(self.audio_path, \n                             sr=self.sr, \n                             mono=self.mono, \n                             offset=self.offset, \n                             duration=self.duration)\n        return y, sr\n    \n    def apply_filter(self, filter_kind=\"\"):\n        if filter_kind == \"wavelet\":\n            self.y = wavelet_filter(self.y, thresh=0.1, wavelet=\"db4\", level=None)\n        elif filter_kind == \"butter_bandpass\":\n            self.y = butter_bandpass_filter(self.y, lowcut, highcut, sr, order=5)\n        else:\n            return\n        \n    def process_sample(self, display_plots=False, **kwargs) -> None:\n        if self.normalize_signal:\n            y = librosa.util.normalize(y)\n        \n        self.apply_filter(self.filter_kind)\n        \n        if display_plots:\n            display_audio(self.y, self.sr, title=f'{self.audio_path}, sr={self.sr}')\n            plot_spectrograms(self.y, self.sr, self.n_fft, self.hop_length, self.n_mels, self.n_mfcc)\n            \n        if self.save_stft:\n            Sxx_db, P = calculate_stft(self.y, self.n_fft, self.hop_length)\n            plt.imsave(f'{self.save_dir}/{self.audio_name}_stft.png', Sxx_db, cmap='viridis')\n\n        if self.save_mel:\n            S_mel_db = calculate_mel_spectrogram(self.y, self.sr, self.n_fft, self.hop_length, self.n_mels)\n            plt.imsave(f'{self.save_dir}/{self.audio_name}_mel.png', S_mel_db, cmap='viridis')\n\n        if self.save_mfcc:\n            S_mfcc = calculate_mfcc(self.y, self.sr, self.n_mfcc)\n            plt.imsave(f'{self.save_dir}/{self.audio_name}_mfcc.png', S_mfcc, cmap='viridis')\n            \n# audio_path = \"/kaggle/input/birdclef-2024/train_audio/zitcis1/XC583640.ogg\"\n# processor = AudioProcessor(audio_path,\n#                           sample_rate=32000)\n\n# processor.process_sample(display_plots=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:08:18.561689Z","iopub.execute_input":"2024-04-23T12:08:18.562165Z","iopub.status.idle":"2024-04-23T12:08:18.580367Z","shell.execute_reply.started":"2024-04-23T12:08:18.562132Z","shell.execute_reply":"2024-04-23T12:08:18.579111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metadata EDA","metadata":{}},{"cell_type":"code","source":"metadata_df = pd.read_csv(\"/kaggle/input/birdclef-2024/train_metadata.csv\")\nmetadata_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:08:18.581802Z","iopub.execute_input":"2024-04-23T12:08:18.582229Z","iopub.status.idle":"2024-04-23T12:08:18.733324Z","shell.execute_reply.started":"2024-04-23T12:08:18.582189Z","shell.execute_reply":"2024-04-23T12:08:18.731856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of train samples:\", metadata_df.shape[0])\nprint(\"Number of unlabeled samples:\", len(os.listdir('/kaggle/input/birdclef-2024/unlabeled_soundscapes')))\nprint(\"\\nMissing values:\\n\", metadata_df.isnull().sum())\nprint(\"\\nDuplicate rows:\\n\", metadata_df[metadata_df.duplicated()])\nprint(\"\\nSummary statistics for numerical variables:\\n\", metadata_df.describe())\nprint(\"\\nPrimary label values count:\\n\", metadata_df['primary_label'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:08:18.735927Z","iopub.execute_input":"2024-04-23T12:08:18.736290Z","iopub.status.idle":"2024-04-23T12:08:18.820230Z","shell.execute_reply.started":"2024-04-23T12:08:18.736258Z","shell.execute_reply":"2024-04-23T12:08:18.818543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## primary_label column","metadata":{}},{"cell_type":"code","source":"primary_label_counts = metadata_df.primary_label.value_counts().sort_values()\n\nfig = px.bar(\n    x=primary_label_counts.values, y=primary_label_counts.index,\n    color_discrete_sequence=['cornflowerblue'],\n    orientation='h', height=1000\n)\nfig.update_layout(\n    title=\"Bar Plot of primary_label\",\n    xaxis_title=\"Count\", yaxis_title=\"Label\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:08:18.822543Z","iopub.execute_input":"2024-04-23T12:08:18.822904Z","iopub.status.idle":"2024-04-23T12:08:18.908108Z","shell.execute_reply.started":"2024-04-23T12:08:18.822874Z","shell.execute_reply":"2024-04-23T12:08:18.906514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top5 = list(primary_label_counts.index[-5:])[::-1]\ntop5","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:08:18.909879Z","iopub.execute_input":"2024-04-23T12:08:18.911053Z","iopub.status.idle":"2024-04-23T12:08:18.923193Z","shell.execute_reply.started":"2024-04-23T12:08:18.910989Z","shell.execute_reply":"2024-04-23T12:08:18.921750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## secondary_labels column","metadata":{}},{"cell_type":"code","source":"print(\"Number of empty secondary column occurrences:\", sum(metadata_df.secondary_labels == \"[]\"))\n\nsecondary_labels_flat = sum([eval(x) for x in metadata_df.secondary_labels], [])\nsecondary_labels_flat_counts = pd.Series(secondary_labels_flat).value_counts().sort_values()\n\nfig = px.bar(\n    x=secondary_labels_flat_counts.values, y=secondary_labels_flat_counts.index,\n    color_discrete_sequence=['cornflowerblue'],\n    orientation='h', height=1000\n)\nfig.update_layout(\n    title=\"Bar Plot of secondary_labels\",\n    xaxis_title=\"Count\", yaxis_title=\"Secondary Labels\"\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:08:18.925771Z","iopub.execute_input":"2024-04-23T12:08:18.926201Z","iopub.status.idle":"2024-04-23T12:08:19.384674Z","shell.execute_reply.started":"2024-04-23T12:08:18.926160Z","shell.execute_reply":"2024-04-23T12:08:19.383309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## type","metadata":{}},{"cell_type":"code","source":"print(\"Number of empty type occurrences:\", sum(metadata_df['type'] == \"[]\"))\n\ntype_labels_flat = sum([eval(x) for x in metadata_df['type']], [])\ntype_labels_flat_counts = pd.Series(type_labels_flat).value_counts().sort_values()\n\nfig = px.bar(\n    x=type_labels_flat_counts.values, y=[\n        f'{\" \".join(x.split(\" \")[:3])} ...' if len(x.split(\" \")) > 3 else x \n        for x in type_labels_flat_counts.index\n    ],\n    color_discrete_sequence=['darkgoldenrod'],\n    orientation='h', height=1000\n)\nfig.update_layout(\n    title=\"Bar Plot of type\",\n    xaxis_title=\"Count\", yaxis_title=\"Audio type\")\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:08:19.385888Z","iopub.execute_input":"2024-04-23T12:08:19.386200Z","iopub.status.idle":"2024-04-23T12:08:22.914000Z","shell.execute_reply.started":"2024-04-23T12:08:19.386174Z","shell.execute_reply":"2024-04-23T12:08:22.912703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"top_20_type_counts = type_labels_flat_counts.head(20)\n\nfig = px.bar(\n    x=top_20_type_counts.values, \n    y=[\n        f'{\" \".join(x.split(\" \")[:3])} ...' if len(x.split(\" \")) > 3 else x \n        for x in top_20_type_counts.index\n    ],\n    color_discrete_sequence=['cornflowerblue'],\n    orientation='h', height=1000\n)\nfig.update_layout(\n    title=\"Bar Plot of Top 20 Audio Types\",\n    xaxis_title=\"Count\", yaxis_title=\"Type\"\n)\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:08:22.915247Z","iopub.execute_input":"2024-04-23T12:08:22.915608Z","iopub.status.idle":"2024-04-23T12:08:22.983948Z","shell.execute_reply.started":"2024-04-23T12:08:22.915575Z","shell.execute_reply":"2024-04-23T12:08:22.982510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## latitude, longitude columns","metadata":{}},{"cell_type":"code","source":"no_nan_metadata_df = metadata_df.dropna()\n\nfig = px.scatter_mapbox(\n    no_nan_metadata_df, \n    lat=\"latitude\", lon=\"longitude\", color=\"common_name\",\n    hover_name=\"filename\", hover_data=[\"common_name\", \"author\", \"rating\"],\n    zoom=2, height=500, \n    center=dict(\n        lat=no_nan_metadata_df['latitude'].mean(), \n        lon=no_nan_metadata_df['longitude'].mean()\n    )\n)\n\nfig.update_layout(mapbox_style=\"open-street-map\")\nfig.update_layout(margin={\"r\":0,\"t\":0,\"l\":0,\"b\":0})\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:08:22.985509Z","iopub.execute_input":"2024-04-23T12:08:22.985879Z","iopub.status.idle":"2024-04-23T12:08:24.529417Z","shell.execute_reply.started":"2024-04-23T12:08:22.985851Z","shell.execute_reply":"2024-04-23T12:08:24.525889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"no_nan_metadata_top5_df = no_nan_metadata_df[no_nan_metadata_df['primary_label'].isin(top5)]\n\nfig = px.scatter_mapbox(\n    no_nan_metadata_top5_df, \n    lat=\"latitude\", lon=\"longitude\", color=\"common_name\",\n    hover_name=\"filename\", hover_data=[\"common_name\", \"author\", \"rating\"],\n    zoom=2, height=500, \n    center=dict(\n        lat=no_nan_metadata_top5_df['latitude'].mean(), \n        lon=no_nan_metadata_top5_df['longitude'].mean()\n    )\n)\n\nfig.update_layout(mapbox_style=\"open-street-map\")\nfig.update_layout(margin={\"r\":0,\"t\":0,\"l\":0,\"b\":0})\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:08:24.531276Z","iopub.execute_input":"2024-04-23T12:08:24.531704Z","iopub.status.idle":"2024-04-23T12:08:24.699591Z","shell.execute_reply.started":"2024-04-23T12:08:24.531670Z","shell.execute_reply":"2024-04-23T12:08:24.698534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Audio EDA","metadata":{}},{"cell_type":"code","source":"train_audio_dir = '/kaggle/input/birdclef-2024/train_audio/'\n\nfor label in top5:\n    files = glob.glob(os.path.join(train_audio_dir, f'*{label}/*.ogg'), recursive=True)\n    \n    example_file = random.choice(files) if files else None\n    \n    if example_file:\n        example_file = files[0]\n        processor = AudioProcessor(example_file,\n                                   sample_rate=32000)\n        \n        processor.process_sample(display_plots=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:08:24.700984Z","iopub.execute_input":"2024-04-23T12:08:24.701312Z","iopub.status.idle":"2024-04-23T12:08:50.087335Z","shell.execute_reply.started":"2024-04-23T12:08:24.701283Z","shell.execute_reply":"2024-04-23T12:08:50.085406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Audio EDA","metadata":{}},{"cell_type":"code","source":"test_audio_dir = '/kaggle/input/birdclef-2024/unlabeled_soundscapes'\n\nfor _ in range(2):\n    files = glob.glob(os.path.join(test_audio_dir, f'*.ogg'), recursive=True)\n    \n    example_file = random.choice(files) if files else None\n    \n    if example_file:\n        example_file = files[0]\n        processor = AudioProcessor(example_file,\n                                   sample_rate=32000)\n        \n        processor.process_sample(display_plots=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:08:50.091355Z","iopub.execute_input":"2024-04-23T12:08:50.091770Z","iopub.status.idle":"2024-04-23T12:09:27.430742Z","shell.execute_reply.started":"2024-04-23T12:08:50.091739Z","shell.execute_reply":"2024-04-23T12:09:27.429177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Audio duration analysis\n\nPreparing audio data, including raw waveforms and spectrogram images, for training and inference typically involves several preprocessing steps:\n\n1. **Resampling:** Ensuring a consistent sample rate across all data samples. Since the test data sample rate is 32,000 Hz, it's crucial to load waveforms for the training stage with the appropriate sample rate.\n\n2. **Channel Consistency:** Aligning the number of channels with the input requirements of the corresponding deep learning model.\n\n3. **Waveform Length Truncation:** Adjusting the length of audio waveforms to a standardized duration. \n\nSince signals in the train dataset have different durations, analyzing audio duration statistics is necessary to determine the optimal length. Additionally, identifying the first peak of amplitude will help in selecting an optimal start point, avoiding unnecessary silence at the beginning of the audio.","metadata":{}},{"cell_type":"code","source":"train_audio_dir = '/kaggle/input/birdclef-2024/train_audio/'\nfiles = glob.glob(os.path.join(train_audio_dir, '**/*.ogg'), recursive=True)\nfiles = sorted(files)\n\n# extracts duration in seconds\ndurations = [librosa.get_duration(path=file, sr=32000) for file in tqdm(files)]","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:09:27.432541Z","iopub.execute_input":"2024-04-23T12:09:27.432948Z","iopub.status.idle":"2024-04-23T12:11:07.233013Z","shell.execute_reply.started":"2024-04-23T12:09:27.432913Z","shell.execute_reply":"2024-04-23T12:11:07.231880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df['duration_seconds'] = durations\nduration_stats = metadata_df['duration_seconds'].describe()\nprint(duration_stats)\n\nfig_hist = px.histogram(metadata_df, x='duration_seconds', nbins=30, title='Audio Duration (Seconds) Histogram')\nfig_hist.show()\n\nfig_box = px.box(metadata_df, y='duration_seconds', title='Audio Duration (Seconds) Box Plot')\nfig_box.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:11:07.235027Z","iopub.execute_input":"2024-04-23T12:11:07.235531Z","iopub.status.idle":"2024-04-23T12:11:07.439909Z","shell.execute_reply.started":"2024-04-23T12:11:07.235486Z","shell.execute_reply":"2024-04-23T12:11:07.438204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df['duration_minutes'] = metadata_df['duration_seconds'] / 60\nduration_stats_minutes = metadata_df['duration_minutes'].describe()\nprint(duration_stats_minutes)\n\nfig_hist = px.histogram(metadata_df, x='duration_minutes', nbins=30, title='Audio Duration (Minutes) Histogram')\nfig_hist.show()\n\nfig_box = px.box(metadata_df, y='duration_minutes', title='Audio Duration (Minutes) Box Plot')\nfig_box.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:11:07.441785Z","iopub.execute_input":"2024-04-23T12:11:07.442161Z","iopub.status.idle":"2024-04-23T12:11:07.655156Z","shell.execute_reply.started":"2024-04-23T12:11:07.442131Z","shell.execute_reply":"2024-04-23T12:11:07.653362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on the statistics, it seems that the majority of audio files have durations less than 1 minute, with a mean duration of approximately 0.70 minutes (or 42 seconds). The median duration is around 0.37 minutes (or 22 seconds).\n\nHowever, there are some outliers with much longer durations, as indicated by the maximum duration of 99.4 minutes.\n\nAlso, the minimum duration in the dataset is approximately 0.007833 minutes, which is less than 1 second.","metadata":{}},{"cell_type":"code","source":"filtered_metadata_df = metadata_df[metadata_df['duration_seconds'] < 2]\nprint(\"Number of files less than 2 seconds:\", len(filtered_metadata_df))\n\n# investigate files with short duration\ntrain_audio_dir = '/kaggle/input/birdclef-2024/train_audio/'\naudio_path = os.path.join(train_audio_dir, filtered_metadata_df['filename'].iloc[1])\n\ny, sr = load_audio(audio_path)\ndisplay_audio(y, sr, title=audio_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:11:07.657240Z","iopub.execute_input":"2024-04-23T12:11:07.657766Z","iopub.status.idle":"2024-04-23T12:11:08.247398Z","shell.execute_reply.started":"2024-04-23T12:11:07.657728Z","shell.execute_reply":"2024-04-23T12:11:08.245489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some of files less than 2 seconds contains background noise.","metadata":{}},{"cell_type":"code","source":"filtered_metadata_df = metadata_df[metadata_df['duration_minutes'] > 20]\nprint(\"Number of files longer than 20 minutes:\", len(filtered_metadata_df))\n\n# investigate files with long duration\ntrain_audio_dir = '/kaggle/input/birdclef-2024/train_audio/'\naudio_path = os.path.join(train_audio_dir, filtered_metadata_df['filename'].iloc[1])\n\ny, sr = load_audio(audio_path, duration=20.0)\ndisplay_audio(y, sr, title=audio_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:11:08.249465Z","iopub.execute_input":"2024-04-23T12:11:08.249986Z","iopub.status.idle":"2024-04-23T12:11:08.887343Z","shell.execute_reply.started":"2024-04-23T12:11:08.249943Z","shell.execute_reply":"2024-04-23T12:11:08.885676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So, to find optimal duration, let's calculate the duration at a certain percentile, such as 95th or 99th percentile.","metadata":{}},{"cell_type":"code","source":"percentile_1 = np.percentile(metadata_df['duration_minutes'].values, 1)\nprint(f\"Duration at 1st percentile: {percentile_1} minutes\")\n\npercentile_95 = np.percentile(metadata_df['duration_minutes'].values, 95)\nprint(f\"Duration at 95th percentile: {percentile_95} minutes\")\n\npercentile_99 = np.percentile(metadata_df['duration_minutes'].values, 99)\nprint(f\"Duration at 99th percentile: {percentile_99} minutes\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:11:08.889663Z","iopub.execute_input":"2024-04-23T12:11:08.890212Z","iopub.status.idle":"2024-04-23T12:11:08.905450Z","shell.execute_reply.started":"2024-04-23T12:11:08.890164Z","shell.execute_reply":"2024-04-23T12:11:08.904156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing: noise removal\n\nSome data samples, such as \"/kaggle/input/birdclef-2024/train_audio/zitcis1/XC583640.ogg\" contain background noise, while others may contain overlapping bird vocalizations (\"/kaggle/input/birdclef-2024/train_audio/eucdov/XC112763.ogg\"). Therefore, as an additional stage of preprocessing, it is crucial to implement noise filtering techniques.\n\nLet's investigate some noise removal techniques and compare audios before and after noise removal.\n\nSome of filtering techniques involve:\n- FIR/IIR filters;\n- Low-pass, high-pass, bandpass filters;\n- wavelet-based filtering;\n- median filter;\n- adaptive filter;\n\nAlso, it could be useful to analyze silence section at the beginning of the signal to estimate noise reference signal, which could help to estimate SNR and apply other filtering methods which use noise as a reference (spectral substraction, Wiener filtering, and etc).","metadata":{}},{"cell_type":"markdown","source":"## Wavelet-based filtering","metadata":{}},{"cell_type":"code","source":"audio_path = \"/kaggle/input/birdclef-2024/train_audio/zitcis1/XC583640.ogg\"\ny, sr = load_audio(audio_path)\ndenoised_signal = wavelet_filter(y)\ndisplay_audio(y, sr, title=f\"Original Signal\\n{audio_path}\")\ndisplay_audio(denoised_signal, sr, title=f\"Denoised Signal\\n{audio_path}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:11:08.907270Z","iopub.execute_input":"2024-04-23T12:11:08.908065Z","iopub.status.idle":"2024-04-23T12:11:10.222906Z","shell.execute_reply.started":"2024-04-23T12:11:08.908018Z","shell.execute_reply":"2024-04-23T12:11:10.221537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After applying wavelet filtering, the denoised signal exhibits a significant reduction in noise levels, indicating that high-frequency noise components have been attenuated. However, the amplitude of the bird vocalization was also slightly reduced.\n\nIt is interesting to play with truncation threshold, algorithm to estimate optimal truncation threshold, wavelet basis function (I used Daubechies), and levels of decomposition to get desired outcome. In this sense, wavelet filtering is more automated in comparison to band-based filters, where specific frequency bands of interest must be manually selected, which can vary from sample to sample","metadata":{}},{"cell_type":"code","source":"n = len(y)\nfreq = np.fft.fftfreq(n, d=1/sr)\n\nfft_original = np.fft.fft(y)\nfft_original_mag = np.abs(fft_original)\n\nfft_denoised = np.fft.fft(denoised_signal)\nfft_denoised_mag = np.abs(fft_denoised)\n\nplt.figure(figsize=(12, 6))\nplt.subplot(2, 1, 1)\nplt.plot(freq[:n//2], fft_original_mag[:n//2], color='blue')\nplt.title('Fourier Spectrum of Original Signal')\nplt.xlabel('Frequency (Hz)')\nplt.ylabel('Magnitude')\n\nplt.subplot(2, 1, 2)\nplt.plot(freq[:n//2], fft_denoised_mag[:n//2], color='red')\nplt.title('Fourier Spectrum of Denoised Signal')\nplt.xlabel('Frequency (Hz)')\nplt.ylabel('Magnitude')\n\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-23T12:11:10.224778Z","iopub.execute_input":"2024-04-23T12:11:10.225236Z","iopub.status.idle":"2024-04-23T12:11:12.776649Z","shell.execute_reply.started":"2024-04-23T12:11:10.225200Z","shell.execute_reply":"2024-04-23T12:11:12.775010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"S_mel_db_original = calculate_mel_spectrogram(y, sr=32000, n_fft=1024, hop_length=512, n_mels=256)\nS_mel_db_denoised = calculate_mel_spectrogram(denoised_signal, sr=32000, n_fft=1024, hop_length=512, n_mels=128)\n\nplt.figure(figsize=(14, 4))\nplt.subplot(1, 2, 1)\nlibrosa.display.specshow(S_mel_db_original, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"Original Signal Mel Spectrogram\")\n\nplt.subplot(1, 2, 2)\nlibrosa.display.specshow(S_mel_db_denoised, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"Denoised Signal (Wavelet) Mel Spectrogram\")\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-04-23T12:11:12.778334Z","iopub.execute_input":"2024-04-23T12:11:12.778736Z","iopub.status.idle":"2024-04-23T12:11:14.739962Z","shell.execute_reply.started":"2024-04-23T12:11:12.778703Z","shell.execute_reply":"2024-04-23T12:11:14.738704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's consider another sample with bird's vocalizations overlapping","metadata":{}},{"cell_type":"code","source":"audio_path = \"/kaggle/input/birdclef-2024/train_audio/eucdov/XC112763.ogg\"\ny, sr = load_audio(audio_path)\ndenoised_signal = wavelet_filter(y)\ndisplay_audio(y, sr, title=f\"Original Signal\\n{audio_path}\")\ndisplay_audio(denoised_signal, sr, title=f\"Denoised Signal\\n{audio_path}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:11:14.741903Z","iopub.execute_input":"2024-04-23T12:11:14.742676Z","iopub.status.idle":"2024-04-23T12:11:15.954631Z","shell.execute_reply.started":"2024-04-23T12:11:14.742614Z","shell.execute_reply":"2024-04-23T12:11:15.953201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = len(y)\nfreq = np.fft.fftfreq(n, d=1/sr)\n\nfft_original = np.fft.fft(y)\nfft_original_mag = np.abs(fft_original)\n\nfft_denoised = np.fft.fft(denoised_signal)\nfft_denoised_mag = np.abs(fft_denoised)\n\nplt.figure(figsize=(12, 6))\nplt.subplot(2, 1, 1)\nplt.plot(freq[:n//2], fft_original_mag[:n//2], color='blue')\nplt.title('Fourier Spectrum of Original Signal')\nplt.xlabel('Frequency (Hz)')\nplt.ylabel('Magnitude')\n\nplt.subplot(2, 1, 2)\nplt.plot(freq[:n//2], fft_denoised_mag[:n//2], color='red')\nplt.title('Fourier Spectrum of Denoised Signal')\nplt.xlabel('Frequency (Hz)')\nplt.ylabel('Magnitude')\n\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-23T12:11:15.956495Z","iopub.execute_input":"2024-04-23T12:11:15.956975Z","iopub.status.idle":"2024-04-23T12:11:16.972545Z","shell.execute_reply.started":"2024-04-23T12:11:15.956930Z","shell.execute_reply":"2024-04-23T12:11:16.970533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some spectra exhibit long tails, which can be mitigated by applying a windowing function to the frequency domain. One approach to achieve this is to perform FFT on the signal, apply a window to the resulting spectrum, and then inverse iFFT to transform the signal back to the time domain.","metadata":{}},{"cell_type":"code","source":"S_mel_db_original = calculate_mel_spectrogram(y, sr=32000, n_fft=1024, hop_length=512, n_mels=128)\nS_mel_db_denoised = calculate_mel_spectrogram(denoised_signal, sr=32000, n_fft=1024, hop_length=512, n_mels=256)\n\nplt.figure(figsize=(14, 4))\nplt.subplot(1, 2, 1)\nlibrosa.display.specshow(S_mel_db_original, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"Original Signal Mel Spectrogram\")\n\nplt.subplot(1, 2, 2)\nlibrosa.display.specshow(S_mel_db_denoised, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"Denoised Signal (Wavelet) Mel Spectrogram\")\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-23T12:11:16.974601Z","iopub.execute_input":"2024-04-23T12:11:16.975131Z","iopub.status.idle":"2024-04-23T12:11:18.208590Z","shell.execute_reply.started":"2024-04-23T12:11:16.975087Z","shell.execute_reply":"2024-04-23T12:11:18.207333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bandpass filter","metadata":{}},{"cell_type":"code","source":"audio_path = \"/kaggle/input/birdclef-2024/train_audio/zitcis1/XC583640.ogg\"\ny, sr = load_audio(audio_path)\n\nlowcut, highcut = 5000, 8000 # Hz\n\ndenoised_signal = butter_bandpass_filter(y, lowcut, highcut, sr)\n\ndisplay_audio(y, sr, title=f\"Original Signal\\n{audio_path}\")\ndisplay_audio(denoised_signal, sr, title=f\"Denoised Signal\\n{audio_path} (Bandpass: {lowcut}-{highcut} Hz)\")","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:11:18.210234Z","iopub.execute_input":"2024-04-23T12:11:18.211307Z","iopub.status.idle":"2024-04-23T12:11:19.547465Z","shell.execute_reply.started":"2024-04-23T12:11:18.211269Z","shell.execute_reply":"2024-04-23T12:11:19.546064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = len(y)\nfreq = np.fft.fftfreq(n, d=1/sr)\n\nfft_original = np.fft.fft(y)\nfft_original_mag = np.abs(fft_original)\n\nfft_denoised = np.fft.fft(denoised_signal)\nfft_denoised_mag = np.abs(fft_denoised)\n\nplt.figure(figsize=(12, 6))\nplt.subplot(2, 1, 1)\nplt.plot(freq[:n//2], fft_original_mag[:n//2], color='blue')\nplt.title('Fourier Spectrum of Original Signal')\nplt.xlabel('Frequency (Hz)')\nplt.ylabel('Magnitude')\n\nplt.subplot(2, 1, 2)\nplt.plot(freq[:n//2], fft_denoised_mag[:n//2], color='red')\nplt.title('Fourier Spectrum of Denoised Signal')\nplt.xlabel('Frequency (Hz)')\nplt.ylabel('Magnitude')\n\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-04-23T12:11:19.548746Z","iopub.execute_input":"2024-04-23T12:11:19.549084Z","iopub.status.idle":"2024-04-23T12:11:22.647345Z","shell.execute_reply.started":"2024-04-23T12:11:19.549053Z","shell.execute_reply":"2024-04-23T12:11:22.645333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The low cut-off frequency and the high cut-off frequency in band-pass filters are hyperparameters that need to be adjusted for every sample, as the frequency ranges of interest may vary. An alternative approach is to utilize the built-in filtering functionality provided by **librosa**, which operates by dividing the signal spectrum into distinct frequency bands.","metadata":{}},{"cell_type":"code","source":"S_mel_db_original = calculate_mel_spectrogram(y, sr=32000, n_fft=1024, hop_length=512, n_mels=256)\nS_mel_db_denoised = calculate_mel_spectrogram(denoised_signal, sr=32000, n_fft=1024, hop_length=512, n_mels=256)\n\nplt.figure(figsize=(14, 4))\nplt.subplot(1, 2, 1)\nlibrosa.display.specshow(S_mel_db_original, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"Original Signal Mel Spectrogram\")\n\nplt.subplot(1, 2, 2)\nlibrosa.display.specshow(S_mel_db_denoised, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar(format='%+2.0f dB')\nplt.title(\"Denoised Signal (Wavelet) Mel Spectrogram\")\nplt.tight_layout()\nplt.show()","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-04-23T12:11:22.649537Z","iopub.execute_input":"2024-04-23T12:11:22.650202Z","iopub.status.idle":"2024-04-23T12:11:25.091069Z","shell.execute_reply.started":"2024-04-23T12:11:22.650135Z","shell.execute_reply":"2024-04-23T12:11:25.089643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The low cut-off frequency and the high cut-off frequency in band-pass filters are hyperparameters that need to be adjusted for every sample, as the frequency ranges of interest may vary. ","metadata":{}},{"cell_type":"markdown","source":"# Save features","metadata":{}},{"cell_type":"code","source":"train_audio_dir = '/kaggle/input/birdclef-2024/train_audio/'\n\nfiles = glob.glob(os.path.join(train_audio_dir, '**/*.ogg'), recursive=True)\n\nfor file in tqdm(files[:5]):\n    processor = AudioProcessor(file,\n                               sample_rate=32000)\n    processor.process_sample(save_mel=True, duration=360)","metadata":{"execution":{"iopub.status.busy":"2024-04-23T12:11:59.892304Z","iopub.execute_input":"2024-04-23T12:11:59.894133Z","iopub.status.idle":"2024-04-23T12:12:00.561844Z","shell.execute_reply.started":"2024-04-23T12:11:59.894076Z","shell.execute_reply":"2024-04-23T12:12:00.560413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Thanks for your upvotes 🤗","metadata":{}}]}