{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BIRDCLEF 2024: Spectrogram Feature Extraction 📊🎶\n\nThis notebook outlines the process of feature extraction from audio files for the purpose of classifying bird species 🐦. The dataset consists of audio recordings of various bird calls, and the goal is to extract meaningful features that can be used to train a machine learning model to accurately identify different species.\n\n## Objective 🎯\nThe objective in this notebook is to:\n1. Load the audio files 📂.\n2. Visualize audio data 📉.\n3. Perform feature extraction, focusing on the generation of mel spectrograms, and then extracting harmonic and percussive components of this spectrogram 🎵.\n4. Store these spectrograms into folders for model training 🗂️.\n","metadata":{}},{"cell_type":"code","source":"import os\nimport librosa\nfrom PIL import Image\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport IPython.display as ipd\nfrom tqdm import tqdm\nimport matplotlib.cm as cm\nfrom PIL import Image, ImageOps","metadata":{"execution":{"iopub.status.busy":"2024-05-14T16:52:00.813772Z","iopub.execute_input":"2024-05-14T16:52:00.814199Z","iopub.status.idle":"2024-05-14T16:52:00.820098Z","shell.execute_reply.started":"2024-05-14T16:52:00.814169Z","shell.execute_reply":"2024-05-14T16:52:00.818971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CONFIG:\n    # Mel Spec:\n    SHAPE = (256, 256)\n    FRAME_SIZE = 2048 # n_fft\n    AUDIO_DURATION = 5 # 5 sec\n    N_MELS = SHAPE[0]\n    MIN_SAMPLES = 100\n    \n    # Directory:\n    INPUT_DIR = \"/kaggle/input/birdclef-2024/\"\n    OUTPUT_DIR = \"/kaggle/working/features/\"\n    METADATA = os.path.join(INPUT_DIR, \"train_metadata.csv\")\n    TRAIN_AUDIO = os.path.join(INPUT_DIR, \"train_audio\")","metadata":{"execution":{"iopub.status.busy":"2024-05-14T16:52:00.836299Z","iopub.execute_input":"2024-05-14T16:52:00.836650Z","iopub.status.idle":"2024-05-14T16:52:00.843100Z","shell.execute_reply.started":"2024-05-14T16:52:00.836624Z","shell.execute_reply":"2024-05-14T16:52:00.842001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(CONFIG.METADATA)\ndata.head(4)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T16:52:00.844742Z","iopub.execute_input":"2024-05-14T16:52:00.845318Z","iopub.status.idle":"2024-05-14T16:52:01.072436Z","shell.execute_reply.started":"2024-05-14T16:52:00.845289Z","shell.execute_reply":"2024-05-14T16:52:01.071311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Loading Audio**\n\nWe will load an example audio file to understand its properties.","metadata":{}},{"cell_type":"code","source":"sample_path = os.path.join(CONFIG.INPUT_DIR, \"train_audio\", data.iloc[100][\"filename\"])\nsample_path","metadata":{"execution":{"iopub.status.busy":"2024-05-14T16:52:01.074227Z","iopub.execute_input":"2024-05-14T16:52:01.074626Z","iopub.status.idle":"2024-05-14T16:52:01.085349Z","shell.execute_reply.started":"2024-05-14T16:52:01.074592Z","shell.execute_reply":"2024-05-14T16:52:01.083800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"signal, sr = librosa.load(sample_path)\nprint('Signal:', signal)\nprint('Shape of signal:', np.shape(signal))\nprint('Sampling Rate (KHz):', sr)\nprint('Duration of audio:', np.shape(signal)[0]/sr)\nplt.plot(signal)\nplt.show()\nipd.Audio(sample_path)","metadata":{"execution":{"iopub.status.busy":"2024-05-14T17:00:12.980231Z","iopub.execute_input":"2024-05-14T17:00:12.980665Z","iopub.status.idle":"2024-05-14T17:00:13.326474Z","shell.execute_reply.started":"2024-05-14T17:00:12.980626Z","shell.execute_reply":"2024-05-14T17:00:13.325245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Mel Spectrogram Generation**\n\nMel Spectrograms are a type of spectrogram where the frequencies are converted to the Mel scale, which is perceptually more relevant for human hearing. This transformation makes Mel spectrograms particularly useful for audio classification tasks like bird call identification.\n\n### Generating and Storing Mel Spectrograms\nWe will generate Mel spectrograms for each audio file and store them in a designated folder for easy access during the model training phase.","metadata":{}},{"cell_type":"code","source":"fft_result = np.fft.fft(signal)\nfreq_bins = np.fft.fftfreq(len(signal))\n\nplt.figure(figsize=(8, 4))\nplt.plot(freq_bins, np.abs(fft_result))\nplt.xlabel('Frequency (Hz)')\nplt.ylabel('Magnitude')\nplt.title('FFT of Signal')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-14T17:38:07.402932Z","iopub.execute_input":"2024-05-14T17:38:07.403327Z","iopub.status.idle":"2024-05-14T17:38:07.826541Z","shell.execute_reply.started":"2024-05-14T17:38:07.403296Z","shell.execute_reply":"2024-05-14T17:38:07.825284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"D = np.abs(librosa.stft(signal, n_fft=CONFIG.FRAME_SIZE, hop_length=512))\nplt.plot(D,  color='blue')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-14T17:12:36.129868Z","iopub.execute_input":"2024-05-14T17:12:36.130962Z","iopub.status.idle":"2024-05-14T17:12:36.883645Z","shell.execute_reply.started":"2024-05-14T17:12:36.130926Z","shell.execute_reply":"2024-05-14T17:12:36.882150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"librosa.display.specshow(D, sr=sr, x_axis='time', y_axis='linear');\nplt.colorbar();","metadata":{"execution":{"iopub.status.busy":"2024-05-14T17:15:24.692151Z","iopub.execute_input":"2024-05-14T17:15:24.692546Z","iopub.status.idle":"2024-05-14T17:15:25.438061Z","shell.execute_reply.started":"2024-05-14T17:15:24.692518Z","shell.execute_reply":"2024-05-14T17:15:25.436872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HOP_LENGTH = 512\nmel_spec = librosa.feature.melspectrogram(y=signal, sr=sr, n_fft=CONFIG.FRAME_SIZE, hop_length=HOP_LENGTH, n_mels=256)\n\n# Convert power spec to dB (log scale)\nmel_spec = librosa.power_to_db(mel_spec, ref=np.max)\n\n# Normalising\nmel_spec -= mel_spec.min()\nmel_spec /= mel_spec.max()\nprint(f\"Minimum value: {mel_spec.min()}, Maximum value: {mel_spec.max()}\")\n\n# Plotting\nplt.figure(figsize=(5, 4))\nlibrosa.display.specshow(mel_spec, sr=sr, hop_length=HOP_LENGTH, x_axis='time', y_axis='mel')\nplt.colorbar(format='%+2.0f dB')\nplt.title(sample_path.split(\"/\")[-2:])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-14T17:17:59.380853Z","iopub.execute_input":"2024-05-14T17:17:59.381230Z","iopub.status.idle":"2024-05-14T17:17:59.845951Z","shell.execute_reply.started":"2024-05-14T17:17:59.381202Z","shell.execute_reply":"2024-05-14T17:17:59.844776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Harmonic and Percussive Component Separation**\nIn sound processing, the audio signal can be decomposed into harmonic and percussive components. The harmonic part typically captures the tonal quality of the sound, while the percussive part captures the rhythm and attack. These decompositions can be very insightful for distinguishing different types of sounds in audio classification tasks.\n","metadata":{}},{"cell_type":"code","source":"harmonic, percussive = librosa.effects.hpss(signal)\n\nharmonic_mel_spec = librosa.feature.melspectrogram(y=harmonic, sr=sr, n_fft=CONFIG.FRAME_SIZE, hop_length=HOP_LENGTH, n_mels=256)\nharmonic_mel_spec_db = librosa.power_to_db(harmonic_mel_spec, ref=np.max)\nharmonic_mel_spec_db -= harmonic_mel_spec_db.min()\nharmonic_mel_spec_db /= harmonic_mel_spec_db.max()\n\n# Percussive mel spectrogram\npercussive_mel_spec = librosa.feature.melspectrogram(y=percussive, sr=sr, n_fft=CONFIG.FRAME_SIZE, hop_length=HOP_LENGTH, n_mels=256)\npercussive_mel_spec_db = librosa.power_to_db(percussive_mel_spec, ref=np.max)\npercussive_mel_spec_db -= percussive_mel_spec_db.min()\npercussive_mel_spec_db /= percussive_mel_spec_db.max()\n\nplt.figure(figsize=(6, 6))\nplt.subplot(3, 1, 1)\nlibrosa.display.specshow(mel_spec, sr=sr, hop_length=HOP_LENGTH, x_axis='time', y_axis='mel')\nplt.colorbar(format='%+2.0f dB')\nplt.title(sample_path.split(\"/\")[-2:])\n\n# Plot harmonic component\nplt.subplot(3, 1, 2)  # 2 rows, 1 column, 1st subplot\nlibrosa.display.specshow(harmonic_mel_spec_db, sr=sr, hop_length=HOP_LENGTH, x_axis='time', y_axis='mel')\nplt.colorbar(format='%+2.0f dB')\nplt.title('Harmonic Component')\n\n# Plot percussive component\nplt.subplot(3, 1, 3)  # 2 rows, 1 column, 2nd subplot\nlibrosa.display.specshow(percussive_mel_spec_db, sr=sr, hop_length=HOP_LENGTH, x_axis='time', y_axis='mel')\nplt.colorbar(format='%+2.0f dB')\nplt.title('Percussive Component')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-14T17:20:25.372687Z","iopub.execute_input":"2024-05-14T17:20:25.373102Z","iopub.status.idle":"2024-05-14T17:20:27.232918Z","shell.execute_reply.started":"2024-05-14T17:20:25.373072Z","shell.execute_reply":"2024-05-14T17:20:27.231839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Generating Mel Spectrogram and its components**","metadata":{}},{"cell_type":"code","source":"# Creating folders\nfrom pathlib import Path\n\nTMP_DIR = Path('/kaggle/working/features')\nTMP_DIR.mkdir(exist_ok=True)\n# TMP_DIR = Path('/kaggle/working/features/mel_training')\n# TMP_DIR.mkdir(exist_ok=True)\n# TMP_DIR = Path('/kaggle/working/features/harmonic_training')\n# TMP_DIR.mkdir(exist_ok=True)\n# TMP_DIR = Path('/kaggle/working/features/percussive_training')\n# TMP_DIR.mkdir(exist_ok=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate Mel spec and save given the filename:\ndef save_mel_spec(filename, species, chunk, chunk_no, sr, HOP_LENGTH):\n    mel_spec = librosa.feature.melspectrogram(y=chunk, sr=sr, n_fft=CONFIG.FRAME_SIZE, hop_length=HOP_LENGTH, n_mels=CONFIG.SHAPE[0])\n\n    # Convert power spec to dB (log scale)\n    mel_spec = librosa.power_to_db(mel_spec, ref=np.max)\n\n    # Normalising\n    mel_spec -= mel_spec.min()\n    mel_spec /= (mel_spec.max()- mel_spec.min())\n    \n    #set color mapping \n    mel_spec = cm.magma(mel_spec)[:, :, :3]\n    mel_spec = (mel_spec * 255).astype(np.uint8)\n    mel_spec = Image.fromarray(mel_spec)\n        \n    #resize and flip (so consistent with original dataset)\n    mel_spec = mel_spec.resize(CONFIG.SHAPE, Image.Resampling.LANCZOS)\n    mel_spec = ImageOps.flip(mel_spec)\n    \n    \n    output_folder = os.path.join(CONFIG.OUTPUT_DIR, \"mel_training\", species)\n    if not os.path.exists(output_folder):\n        os.makedirs(output_folder)\n        \n    output = os.path.join(output_folder, f\"{filename.split('.')[0]}_{(chunk_no+1)*5}.png\")\n        \n    mel_spec.save(output)    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting harmonic component of Mel spec and save given the filename:\ndef save_harmonic_mel_spec(filename, species, chunk, chunk_no, sr, HOP_LENGTH):\n    harmonic_mel_spec = librosa.feature.melspectrogram(y=chunk, sr=sr, n_fft=CONFIG.FRAME_SIZE, hop_length=HOP_LENGTH, n_mels=CONFIG.SHAPE[0])\n    harmonic_mel_spec_db = librosa.power_to_db(harmonic_mel_spec, ref=np.max)\n    harmonic_mel_spec_db -= harmonic_mel_spec_db.min()\n    harmonic_mel_spec_db /= (harmonic_mel_spec_db.max() -  harmonic_mel_spec_db.min())\n    \n    #set color mapping \n    harmonic_mel_spec = cm.magma(harmonic_mel_spec_db)[:, :, :3]\n    harmonic_mel_spec = (harmonic_mel_spec * 255).astype(np.uint8)\n    harmonic_mel_spec = Image.fromarray(harmonic_mel_spec)\n        \n    #resize and flip (so consistent with original dataset)\n    harmonic_mel_spec = harmonic_mel_spec.resize(CONFIG.SHAPE, Image.Resampling.LANCZOS)\n    harmonic_mel_spec = ImageOps.flip(harmonic_mel_spec)\n    \n    output_folder = os.path.join(CONFIG.OUTPUT_DIR, \"harmonic_training\", species)\n    if not os.path.exists(output_folder):\n        os.makedirs(output_folder)\n        \n    output = os.path.join(output_folder, f\"{filename.split('.')[0]}_{(chunk_no+1)*5}.png\")\n        \n    harmonic_mel_spec.save(output)    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Extracting percussive component from Mel spec and save given the filename:\ndef save_percussive_mel_spec(filename, species, chunk, chunk_no, sr, HOP_LENGTH):\n    percussive_mel_spec = librosa.feature.melspectrogram(y=chunk, sr=sr, n_fft=CONFIG.FRAME_SIZE, hop_length=HOP_LENGTH, n_mels=CONFIG.SHAPE[0])\n    percussive_mel_spec_db = librosa.power_to_db(percussive_mel_spec, ref=np.max)\n    percussive_mel_spec_db -= percussive_mel_spec_db.min()\n    percussive_mel_spec_db /= (percussive_mel_spec_db.max() - percussive_mel_spec_db.min())\n    \n    #set color mapping \n    percussive_mel_spec_db = cm.magma(percussive_mel_spec_db)[:, :, :3]\n    percussive_mel_spec_db = (percussive_mel_spec_db * 255).astype(np.uint8)\n    percussive_mel_spec_db = Image.fromarray(percussive_mel_spec_db)\n        \n    #resize and flip (so consistent with original dataset)\n    percussive_mel_spec_db = percussive_mel_spec_db.resize(CONFIG.SHAPE, Image.Resampling.LANCZOS)\n    percussive_mel_spec_db = ImageOps.flip(percussive_mel_spec_db)\n    \n    \n    output_folder = os.path.join(CONFIG.OUTPUT_DIR, \"percussive_training\", species)\n    if not os.path.exists(output_folder):\n        os.makedirs(output_folder)\n        \n    output = os.path.join(output_folder, f\"{filename.split('.')[0]}_{(chunk_no+1)*5}.png\")\n        \n    percussive_mel_spec_db.save(output)    ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Parse through the input files \ndef get_features():\n    for species in os.listdir(CONFIG.TRAIN_AUDIO):\n        \n        species_folder = os.path.join(CONFIG.TRAIN_AUDIO, species)\n        file_count = 0\n        for filename in tqdm(os.listdir(species_folder), total=min(CONFIG.MIN_SAMPLES, len(os.listdir(species_folder))), desc=f\"Extracting Features for {species}\", leave=False):\n            file_count+=1\n            \n            # Load File\n            signal, sr = librosa.load(os.path.join(species_folder, filename))\n            HOP_LENGTH = int((5*sr)/(CONFIG.SHAPE[1]-1))\n            \n            # Split into chunks\n            splits = []\n            for i in range(0, len(signal), 5*sr):\n                split = signal[i:i + 5*sr]\n                if len(split) < 5*sr: break    \n                splits.append(split)\n\n            # 3) Generating a mel spectrogram for each 5s chunk\n            for i, chunk in enumerate(splits):\n                if i>3: break\n                _, percussive = librosa.effects.hpss(chunk)\n#                 save_harmonic_mel_spec(filename, species, harmonic, i, sr, HOP_LENGTH)\n#                 save_mel_spec(filename, species, chunk, i, sr, HOP_LENGTH)\n                save_percussive_mel_spec(filename, species, percussive, i, sr, HOP_LENGTH)\n                \n            \n            # 4) Taking only first 100 samples\n            if file_count>=CONFIG.MIN_SAMPLES: break\n        ","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get_features()\nprint(\"Extracting Features Completed Successfully!\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}