{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-19T04:33:22.520526Z","iopub.execute_input":"2025-04-19T04:33:22.520781Z","iopub.status.idle":"2025-04-19T04:33:23.021922Z","shell.execute_reply.started":"2025-04-19T04:33:22.520761Z","shell.execute_reply":"2025-04-19T04:33:23.020944Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1. Import Libraries:","metadata":{}},{"cell_type":"code","source":"import os\nimport librosa\nimport librosa.display\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T04:34:32.726865Z","iopub.execute_input":"2025-04-19T04:34:32.727196Z","iopub.status.idle":"2025-04-19T04:34:32.732009Z","shell.execute_reply.started":"2025-04-19T04:34:32.727176Z","shell.execute_reply":"2025-04-19T04:34:32.731153Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 2. load_audio Function:","metadata":{}},{"cell_type":"code","source":"def load_audio(audio_path, target_sr=None, duration=None):\n    \"\"\"Loads an audio file.\n\n    Args:\n        audio_path (str): Path to the audio file.\n        target_sr (int, optional): Target sampling rate. If None, uses the\n            original sampling rate. Defaults to None.\n        duration (float, optional): Target duration in seconds. If the audio\n            is shorter, it will be padded with zeros. If longer, it will be\n            truncated. Defaults to None.\n\n    Returns:\n        tuple: A tuple containing the audio time series (numpy array) and the\n               sampling rate.\n    \"\"\"\n    try:\n        y, sr = librosa.load(audio_path, sr=target_sr)\n        if duration is not None:\n            target_samples = int(duration * sr)\n            if len(y) < target_samples:\n                padding = target_samples - len(y)\n                y = np.pad(y, (0, padding), 'constant')\n            elif len(y) > target_samples:\n                y = y[:target_samples]\n        return y, sr\n    except Exception as e:\n        print(f\"Error loading audio file {audio_path}: {e}\")\n        return None, None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T04:34:37.442021Z","iopub.execute_input":"2025-04-19T04:34:37.442779Z","iopub.status.idle":"2025-04-19T04:34:37.450239Z","shell.execute_reply.started":"2025-04-19T04:34:37.442754Z","shell.execute_reply":"2025-04-19T04:34:37.449144Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 3. extract_mel_spectrogram Function:","metadata":{}},{"cell_type":"code","source":"def extract_mel_spectrogram(audio, sr, n_fft=2048, hop_length=512, n_mels=128):\n    \"\"\"Extracts a Mel spectrogram from an audio signal.\n\n    Args:\n        audio (np.ndarray): The audio time series.\n        sr (int): The sampling rate of the audio.\n        n_fft (int, optional): Length of the FFT window. Defaults to 2048.\n        hop_length (int, optional): Number of audio samples between adjacent\n            STFT columns. Defaults to 512.\n        n_mels (int, optional): Number of Mel bands to generate.\n            Defaults to 128.\n\n    Returns:\n        np.ndarray: The Mel spectrogram (shape: (n_mels, time)).\n    \"\"\"\n    if audio is None:\n        return None\n    mel_spectrogram = librosa.feature.melspectrogram(y=audio, sr=sr,\n                                                     n_fft=n_fft,\n                                                     hop_length=hop_length,\n                                                     n_mels=n_mels)\n    return librosa.power_to_db(mel_spectrogram, ref=np.max)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T04:34:47.967747Z","iopub.execute_input":"2025-04-19T04:34:47.968491Z","iopub.status.idle":"2025-04-19T04:34:47.974480Z","shell.execute_reply.started":"2025-04-19T04:34:47.968466Z","shell.execute_reply":"2025-04-19T04:34:47.973647Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 4. process_audio_file Function:","metadata":{}},{"cell_type":"code","source":"def process_audio_file(audio_path, target_sr=32000, duration=5.0,\n                       n_fft=1024, hop_length=512, n_mels=64):\n    \"\"\"Loads an audio file and extracts its Mel spectrogram features.\n\n    Args:\n        audio_path (str): Path to the audio file.\n        target_sr (int, optional): Target sampling rate. Defaults to 32000.\n        duration (float, optional): Target duration in seconds. Defaults to 5.0.\n        n_fft (int, optional): Length of the FFT window. Defaults to 1024.\n        hop_length (int, optional): Number of audio samples between adjacent\n            STFT columns. Defaults to 512.\n        n_mels (int, optional): Number of Mel bands to generate.\n            Defaults to 64.\n\n    Returns:\n        np.ndarray or None: The Mel spectrogram features if loading was\n                             successful, otherwise None.\n    \"\"\"\n    audio, sr = load_audio(audio_path, target_sr=target_sr, duration=duration)\n    if audio is not None:\n        features = extract_mel_spectrogram(audio, sr, n_fft, hop_length, n_mels)\n        return features\n    return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T04:34:53.816956Z","iopub.execute_input":"2025-04-19T04:34:53.817286Z","iopub.status.idle":"2025-04-19T04:34:53.823141Z","shell.execute_reply.started":"2025-04-19T04:34:53.817263Z","shell.execute_reply":"2025-04-19T04:34:53.822153Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 5. process_directory Function:","metadata":{}},{"cell_type":"code","source":"def process_directory(audio_dir, output_dir, target_sr=32000, duration=5.0,\n                      n_fft=1024, hop_length=512, n_mels=64):\n    \"\"\"Processes all audio files in a directory and saves the Mel spectrograms\n    as numpy arrays.\n\n    Args:\n        audio_dir (str): Path to the directory containing audio files.\n        output_dir (str): Path to the directory where the extracted features\n                            will be saved.\n        target_sr (int, optional): Target sampling rate for all audio files.\n            Defaults to 32000.\n        duration (float, optional): Target duration for all audio files in\n            seconds. Defaults to 5.0.\n        n_fft (int, optional): Length of the FFT window. Defaults to 1024.\n        hop_length (int, optional): Number of audio samples between adjacent\n            STFT columns. Defaults to 512.\n        n_mels (int, optional): Number of Mel bands to generate.\n            Defaults to 64.\n    \"\"\"\n    os.makedirs(output_dir, exist_ok=True)\n    for filename in os.listdir(audio_dir):\n        if filename.endswith(('.wav', '.ogg', '.flac', '.mp3')):  # Add more extensions if needed\n            audio_path = os.path.join(audio_dir, filename)\n            features = process_audio_file(audio_path, target_sr, duration,\n                                          n_fft, hop_length, n_mels)\n            if features is not None:\n                name, ext = os.path.splitext(filename)\n                output_path = os.path.join(output_dir, f\"{name}.npy\")\n                np.save(output_path, features)\n                print(f\"Processed and saved features for {filename} to {output_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T04:35:04.043466Z","iopub.execute_input":"2025-04-19T04:35:04.044027Z","iopub.status.idle":"2025-04-19T04:35:04.051042Z","shell.execute_reply.started":"2025-04-19T04:35:04.044002Z","shell.execute_reply":"2025-04-19T04:35:04.050213Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 6. if __name__ == '__main__': Block:","metadata":{}},{"cell_type":"code","source":"if __name__ == '__main__':\n    # Example usage:\n\n    # 1. Process a single audio file\n    audio_file_path = '/kaggle/input/birdclef-2025/train_audio/greani1/XC132190.ogg'  # Replace with the actual path\n    mel_spectrogram = process_audio_file(audio_file_path)\n    if mel_spectrogram is not None:\n        print(\"Mel spectrogram shape for single file:\", mel_spectrogram.shape)\n        # You can now use this 'mel_spectrogram' for further tasks\n\n    # 2. Process all audio files in a directory\n    audio_directory = '/kaggle/input/birdclef-2025/train_audio/'  # Replace with the actual path to your audio directory\n    output_directory = '/kaggle/input/birdclef-2025/train_audio/'  # Replace with the desired output directory\n    process_directory(audio_directory, output_directory)\n    print(f\"Processed all audio files in {audio_directory} and saved features to {output_directory}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T04:35:11.198194Z","iopub.execute_input":"2025-04-19T04:35:11.198482Z","iopub.status.idle":"2025-04-19T04:35:11.261685Z","shell.execute_reply.started":"2025-04-19T04:35:11.198464Z","shell.execute_reply":"2025-04-19T04:35:11.260975Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 7. visualize_mel_spectrogram Function:","metadata":{}},{"cell_type":"code","source":"def visualize_mel_spectrogram(mel_spectrogram_db, sr, title=\"Mel Spectrogram\"):\n    \"\"\"Visualizes the Mel spectrogram.\n\n    Args:\n        mel_spectrogram_db (np.ndarray): Mel spectrogram in dB scale.\n        sr (int): The sampling rate of the audio.\n        title (str, optional): Title of the plot. Defaults to \"Mel Spectrogram\".\n    \"\"\"\n    if mel_spectrogram_db is None:\n        print(\"No Mel spectrogram to visualize.\")\n        return\n\n    plt.figure(figsize=(10, 4))\n    librosa.display.specshow(mel_spectrogram_db, sr=sr, x_axis='time', y_axis='mel')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title(title)\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T04:35:15.006980Z","iopub.execute_input":"2025-04-19T04:35:15.007707Z","iopub.status.idle":"2025-04-19T04:35:15.013484Z","shell.execute_reply.started":"2025-04-19T04:35:15.007680Z","shell.execute_reply":"2025-04-19T04:35:15.012591Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 8. visualize_features Function:","metadata":{}},{"cell_type":"code","source":"def visualize_features(mfccs, chroma, sr, title=\"Audio Features\"):\n    \"\"\"Visualizes MFCCs and Chroma features.\n\n    Args:\n        mfccs (np.ndarray): Mel-frequency cepstral coefficients.\n        chroma (np.ndarray): Chroma features.\n        sr (int): The sampling rate of the audio.\n        title (str, optional): Title of the plot. Defaults to \"Audio Features\".\n    \"\"\"\n    if mfccs is None or chroma is None:\n        print(\"No features to visualize.\")\n        return\n\n    plt.figure(figsize=(12, 6))\n    plt.suptitle(title)\n\n    plt.subplot(2, 1, 1)\n    librosa.display.specshow(mfccs, sr=sr, x_axis='time')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title('MFCCs')\n\n    plt.subplot(2, 1, 2)\n    librosa.display.specshow(chroma, sr=sr, x_axis='time', y_axis='chroma', vmin=0, vmax=1)\n    plt.colorbar()\n    plt.title('Chroma Features')\n    plt.tight_layout(rect=[0, 0.03, 1, 0.95])  # Adjust layout to make space for suptitle\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T04:35:24.166449Z","iopub.execute_input":"2025-04-19T04:35:24.167246Z","iopub.status.idle":"2025-04-19T04:35:24.174060Z","shell.execute_reply.started":"2025-04-19T04:35:24.167221Z","shell.execute_reply":"2025-04-19T04:35:24.173168Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 9. extract_features Function:","metadata":{}},{"cell_type":"code","source":"def extract_features(audio_path, target_sr=None, duration=None, n_fft=2048, hop_length=512, n_mfcc=20, n_chroma=12):\n    \"\"\"Loads an audio file and extracts MFCCs and Chroma features.\n\n    Args:\n        audio_path (str): Path to the audio file.\n        target_sr (int, optional): Target sampling rate. Defaults to None.\n        duration (float, optional): Target duration in seconds. Defaults to None.\n        n_fft (int, optional): Length of the FFT window. Defaults to 2048.\n        hop_length (int, optional): Number of audio samples between adjacent\n            STFT columns. Defaults to 512.\n        n_mfcc (int, optional): Number of Mel-frequency cepstral coefficients\n            to compute. Defaults to 20.\n        n_chroma (int, optional): Number of chroma bins to produce. Defaults to 12.\n\n    Returns:\n        tuple: A tuple containing:\n            - mfccs (np.ndarray): Mel-frequency cepstral coefficients\n              (shape: (n_mfcc, time)).\n            - chroma (np.ndarray): Chroma features (shape: (n_chroma, time)).\n            - sr (int): The sampling rate of the audio.\n    \"\"\"\n    try:\n        y, sr = librosa.load(audio_path, sr=target_sr, duration=duration)\n\n        # Extract MFCCs\n        mfccs = librosa.feature.mfcc(y=y, sr=sr, n_fft=n_fft, hop_length=hop_length, n_mfcc=n_mfcc)\n\n        # Extract Chroma features\n        chroma = librosa.feature.chroma_stft(y=y, sr=sr, n_fft=n_fft, hop_length=hop_length, n_chroma=n_chroma)\n\n        return mfccs, chroma, sr\n\n    except Exception as e:\n        print(f\"Error processing audio file {audio_path}: {e}\")\n        return None, None, None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T04:35:29.181326Z","iopub.execute_input":"2025-04-19T04:35:29.182061Z","iopub.status.idle":"2025-04-19T04:35:29.188470Z","shell.execute_reply.started":"2025-04-19T04:35:29.182036Z","shell.execute_reply":"2025-04-19T04:35:29.187659Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 10. if __name__ == '__main__': Block","metadata":{}},{"cell_type":"code","source":"if __name__ == '__main__':\n    audio_file = '/kaggle/input/birdclef-2025/train_audio/greani1/XC556246.ogg'  # Replace with the actual path to your audio file\n\n    mfccs, chroma, sr = extract_features(audio_file, target_sr=32000, duration=5.0, n_mfcc=20, n_chroma=12)\n\n    if mfccs is not None and chroma is not None:\n        print(\"Shape of MFCCs:\", mfccs.shape)\n        print(\"Shape of Chroma features:\", chroma.shape)\n        print(\"Sampling rate:\", sr)\n\n        visualize_features(mfccs, chroma, sr, title=f\"Features for {audio_file}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-19T04:35:35.292350Z","iopub.execute_input":"2025-04-19T04:35:35.292654Z","iopub.status.idle":"2025-04-19T04:35:36.078860Z","shell.execute_reply.started":"2025-04-19T04:35:35.292631Z","shell.execute_reply":"2025-04-19T04:35:36.077903Z"}},"outputs":[],"execution_count":null}]}