{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":171789695,"sourceType":"kernelVersion"}],"dockerImageVersionId":30474,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Use Librosa to generate some Mel Spectrograms!\n### This version processes /kaggle/input/birdclef-2024/unlabeled_soundscapes","metadata":{}},{"cell_type":"code","source":"import os\n\nimport pandas as pd\nimport numpy as np\n\nimport librosa\nimport librosa.display\n\nfrom IPython.display import Audio\n\nfrom PIL import Image, ImageOps\n\nfrom scipy.signal import butter, filtfilt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-25T21:47:43.477839Z","iopub.execute_input":"2024-04-25T21:47:43.478271Z","iopub.status.idle":"2024-04-25T21:47:43.483986Z","shell.execute_reply.started":"2024-04-25T21:47:43.478235Z","shell.execute_reply":"2024-04-25T21:47:43.482897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# File paths","metadata":{}},{"cell_type":"code","source":"audio_input_folder = \"/kaggle/input/birdclef-2024/unlabeled_soundscapes\"\nimage_output_folder = \"/kaggle/working/\"\nmeta_data = pd.read_csv(\"/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:47:43.485570Z","iopub.execute_input":"2024-04-25T21:47:43.485850Z","iopub.status.idle":"2024-04-25T21:47:43.548381Z","shell.execute_reply.started":"2024-04-25T21:47:43.485824Z","shell.execute_reply":"2024-04-25T21:47:43.547371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Settings","metadata":{}},{"cell_type":"code","source":"#image height (width determined by length) - used for Mels value\nimage_height = 300\n\nimage_width_per_5_seconds = 300\nhop_length = int(160400 / image_width_per_5_seconds)\n\n#bandpass filter for audio (Hz)\nlow_cut = 400\nhigh_cut = 10000","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:47:43.549972Z","iopub.execute_input":"2024-04-25T21:47:43.550299Z","iopub.status.idle":"2024-04-25T21:47:43.555270Z","shell.execute_reply.started":"2024-04-25T21:47:43.550272Z","shell.execute_reply":"2024-04-25T21:47:43.554193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save spectrograms to disk (or preview)","metadata":{}},{"cell_type":"code","source":"def bandpass_filter(data, lowcut, highcut, sr, order=5):\n    nyquist = 0.5 * sr\n    low = lowcut / nyquist\n    high = highcut / nyquist\n    b, a = butter(order, [low, high], btype='band')\n    y = filtfilt(b, a, data)\n    return y\n\n\ndef spectrograms_for_audio(dirname, filename, preview=False):\n    audio_path = os.path.join(dirname, filename)\n    audio_data, sr = librosa.load(audio_path, sr=None)\n    total_duration = librosa.get_duration(path=audio_path)\n    \n    audio_data = bandpass_filter(audio_data, low_cut, high_cut, sr)\n    \n    # Normalize the audio\n    max_value = max(abs(audio_data))\n    scaling_factor = 1.0 / max_value\n    audio_data = audio_data * scaling_factor\n        \n    S = librosa.feature.melspectrogram(y=audio_data, sr=sr, hop_length=hop_length, n_mels=image_height, fmin=low_cut, fmax=high_cut)\n    S_db = librosa.amplitude_to_db(S, ref=np.max)\n\n    normalized_array = (S_db - np.min(S_db)) / (np.max(S_db) - np.min(S_db))\n    spectrogram_image = (normalized_array * 255).astype(np.uint8)  \n    spectrogram_image = Image.fromarray(spectrogram_image) \n\n    if preview == True:\n        display(Audio(data=audio_data, rate=sr))\n        return spectrogram_image\n\n    output_folder = os.path.join(image_output_folder, os.path.basename(os.path.normpath(dirname)))\n    if not os.path.exists(output_folder):\n        os.makedirs(output_folder)\n\n    # Naming files with an offset to indicate seconds offset\n    base_filename = filename.replace('.ogg', '')\n    output_filename = os.path.join(output_folder, f\"{base_filename}.png\")\n\n    spectrogram_image.save(output_filename)\n\n    return total_duration","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:47:43.556475Z","iopub.execute_input":"2024-04-25T21:47:43.556755Z","iopub.status.idle":"2024-04-25T21:47:43.568226Z","shell.execute_reply.started":"2024-04-25T21:47:43.556723Z","shell.execute_reply":"2024-04-25T21:47:43.567279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate the spectrograms!\n### 'X' is displayed for any file not processed due to being shorter than audio_duration","metadata":{}},{"cell_type":"code","source":"# List files in the current directory (in any order)\nfilenames = os.listdir(audio_input_folder)\nfilenames = sorted(filenames)\n\n# Process up to max_audio_input_per_species files\nseconds_audio_this_species = 0\nfor filename in filenames:\n    seconds_audio = spectrograms_for_audio(audio_input_folder, filename)\n    seconds_audio_this_species += seconds_audio\n    print(\".\", end = \"\")\nprint (f\"({seconds_audio_this_species} seconds audio processed)\")","metadata":{"execution":{"iopub.status.busy":"2024-04-25T21:47:43.570227Z","iopub.execute_input":"2024-04-25T21:47:43.570539Z","iopub.status.idle":"2024-04-25T21:48:16.418954Z","shell.execute_reply.started":"2024-04-25T21:47:43.570513Z","shell.execute_reply":"2024-04-25T21:48:16.417466Z"},"trusted":true},"execution_count":null,"outputs":[]}]}