{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30474,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Use Librosa to generate some Mel Spectrograms!\n### Maybe use these to train an ImageNet...\n\n### This version:\n* Each (and every) OGG file gets saved as a single image\n* Height = 224 (same as Mels value)\n* Width = 512 for each 5 seconds\n* Image will need to be cropped / scaled during training\n* Image is saved monochrome\n* Intent is to capture all the data / can down-select later...","metadata":{}},{"cell_type":"code","source":"#!pip install noisereduce\n\nimport os\n\nimport pandas as pd\nimport numpy as np\n\nimport librosa\nimport librosa.display\n\nfrom IPython.display import Audio\n\nfrom PIL import Image, ImageOps\n\nfrom scipy.signal import butter, filtfilt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-13T07:56:50.809595Z","iopub.execute_input":"2024-04-13T07:56:50.810508Z","iopub.status.idle":"2024-04-13T07:56:50.815932Z","shell.execute_reply.started":"2024-04-13T07:56:50.810464Z","shell.execute_reply":"2024-04-13T07:56:50.815151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# File paths","metadata":{}},{"cell_type":"code","source":"audio_input_folder = \"/kaggle/input/birdclef-2024/train_audio\"\nimage_output_folder = \"/kaggle/working/train_images\"\nmeta_data = pd.read_csv(\"/kaggle/input/birdclef-2024/eBird_Taxonomy_v2021.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:56:50.823910Z","iopub.execute_input":"2024-04-13T07:56:50.824801Z","iopub.status.idle":"2024-04-13T07:56:50.899572Z","shell.execute_reply.started":"2024-04-13T07:56:50.824768Z","shell.execute_reply":"2024-04-13T07:56:50.898588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Settings","metadata":{}},{"cell_type":"code","source":"#image height (width determined by length) - used for Mels value\nimage_height = 224\n\n#processes up to this many audio files per species\nmax_seconds_audio_per_species = 1000\n\n#going with a large number (can scale this down in training)\nimage_width_per_5_seconds = 512\nhop_length = int(160400 / image_width_per_5_seconds)\n\n#bandpass filter for audio (Hz)\nlow_cut = 400\nhigh_cut = 10000","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:56:50.901483Z","iopub.execute_input":"2024-04-13T07:56:50.901806Z","iopub.status.idle":"2024-04-13T07:56:50.907975Z","shell.execute_reply.started":"2024-04-13T07:56:50.901778Z","shell.execute_reply":"2024-04-13T07:56:50.906853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Function to lookup bird species","metadata":{}},{"cell_type":"code","source":"def bird_species_from_folder(path):\n    if path.endswith('/'):\n        path = path[:-1]\n\n    last_folder = os.path.basename(path)\n    primary_com_name = meta_data.loc[meta_data['SPECIES_CODE'] == last_folder, 'PRIMARY_COM_NAME']\n    primary_com_name_str = primary_com_name.iloc[0] if not primary_com_name.empty else ''    \n    return f\"{primary_com_name_str} ({last_folder})\"","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:56:50.909064Z","iopub.execute_input":"2024-04-13T07:56:50.909936Z","iopub.status.idle":"2024-04-13T07:56:50.917854Z","shell.execute_reply.started":"2024-04-13T07:56:50.909784Z","shell.execute_reply":"2024-04-13T07:56:50.916814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save spectrograms to disk (or preview)","metadata":{}},{"cell_type":"code","source":"def bandpass_filter(data, lowcut, highcut, sr, order=5):\n    nyquist = 0.5 * sr\n    low = lowcut / nyquist\n    high = highcut / nyquist\n    b, a = butter(order, [low, high], btype='band')\n    y = filtfilt(b, a, data)\n    return y\n\n\ndef spectrograms_for_audio(dirname, filename, preview=False):\n    audio_path = os.path.join(dirname, filename)\n    audio_data, sr = librosa.load(audio_path, sr=None)\n    total_duration = librosa.get_duration(path=audio_path)\n    \n    audio_data = bandpass_filter(audio_data, low_cut, high_cut, sr)\n    \n    # Normalize the audio\n    max_value = max(abs(audio_data))\n    scaling_factor = 1.0 / max_value\n    audio_data = audio_data * scaling_factor\n    \n#    audio_data = remove_silence(audio_data, sr)\n    \n    S = librosa.feature.melspectrogram(y=audio_data, sr=sr, hop_length=hop_length, n_mels=image_height, fmin=low_cut, fmax=high_cut)\n    S_db = librosa.amplitude_to_db(S, ref=np.max)\n\n    normalized_array = (S_db - np.min(S_db)) / (np.max(S_db) - np.min(S_db))\n    spectrogram_image = (normalized_array * 255).astype(np.uint8)  \n    spectrogram_image = Image.fromarray(spectrogram_image) \n\n    if preview == True:\n        display(Audio(data=audio_data, rate=sr))\n        return spectrogram_image\n\n    output_folder = os.path.join(image_output_folder, os.path.basename(os.path.normpath(dirname)))\n    if not os.path.exists(output_folder):\n        os.makedirs(output_folder)\n\n    # Naming files with an offset to indicate seconds offset\n    base_filename = filename.replace('.ogg', '')\n    output_filename = os.path.join(output_folder, f\"{base_filename}.png\")\n\n    spectrogram_image.save(output_filename)\n\n    return total_duration","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:56:50.919641Z","iopub.execute_input":"2024-04-13T07:56:50.919937Z","iopub.status.idle":"2024-04-13T07:56:50.932868Z","shell.execute_reply.started":"2024-04-13T07:56:50.919911Z","shell.execute_reply":"2024-04-13T07:56:50.932169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# A few demos\n","metadata":{}},{"cell_type":"code","source":"spectrograms_for_audio(\"/kaggle/input/birdclef-2024/train_audio/zitcis1/\", \"XC655341.ogg\", preview=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:56:50.933794Z","iopub.execute_input":"2024-04-13T07:56:50.934640Z","iopub.status.idle":"2024-04-13T07:56:51.597921Z","shell.execute_reply.started":"2024-04-13T07:56:50.934601Z","shell.execute_reply":"2024-04-13T07:56:51.596723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"spectrograms_for_audio(\"/kaggle/input/birdclef-2024/train_audio/ashdro1/\", \"XC114599.ogg\", preview=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:56:51.600529Z","iopub.execute_input":"2024-04-13T07:56:51.601113Z","iopub.status.idle":"2024-04-13T07:56:52.256210Z","shell.execute_reply.started":"2024-04-13T07:56:51.601070Z","shell.execute_reply":"2024-04-13T07:56:52.254831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#example of empty spots in clip....\nspectrograms_for_audio(\"/kaggle/input/birdclef-2024/train_audio/asikoe2/\", \"XC190308.ogg\", preview=True)","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:56:52.257705Z","iopub.execute_input":"2024-04-13T07:56:52.258503Z","iopub.status.idle":"2024-04-13T07:56:52.930446Z","shell.execute_reply.started":"2024-04-13T07:56:52.258447Z","shell.execute_reply":"2024-04-13T07:56:52.929332Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate the spectrograms!\n### 'X' is displayed for any file not processed due to being shorter than audio_duration","metadata":{}},{"cell_type":"code","source":"# List everything in the audio input folder\nentries = os.listdir(audio_input_folder)\n\n# Filter to get only directories\nsubdirectories = [entry for entry in entries if os.path.isdir(os.path.join(audio_input_folder, entry))]\n\n# Sort the directories\nsubdirectories.sort()\n\nfolder_count = len(subdirectories)\nfolders_processed = 0\n\n# Iterate through each sorted directory\nfor subdir in subdirectories:\n    subdir_path = os.path.join(audio_input_folder, subdir)\n    folders_processed += 1\n    print(\"\\n\", bird_species_from_folder(subdir_path), folders_processed, \"/\", folder_count)\n\n    # List files in the current directory (in any order)\n    filenames = os.listdir(subdir_path)\n    \n    # Process up to max_audio_input_per_species files\n    seconds_audio_this_species = 0\n    for filename in filenames:\n        seconds_audio = spectrograms_for_audio(subdir_path, filename)\n        seconds_audio_this_species += seconds_audio\n        print(\".\", end = \"\")\n    print (f\"({seconds_audio_this_species} seconds audio processed)\")","metadata":{"execution":{"iopub.status.busy":"2024-04-13T07:56:52.931873Z","iopub.execute_input":"2024-04-13T07:56:52.932640Z","iopub.status.idle":"2024-04-13T07:57:18.433639Z","shell.execute_reply.started":"2024-04-13T07:56:52.932606Z","shell.execute_reply":"2024-04-13T07:57:18.432076Z"},"trusted":true},"execution_count":null,"outputs":[]}]}