{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/birdclef-2021/train_soundscapes'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-12T18:33:55.256662Z","iopub.execute_input":"2023-06-12T18:33:55.257088Z","iopub.status.idle":"2023-06-12T18:33:55.299957Z","shell.execute_reply.started":"2023-06-12T18:33:55.257054Z","shell.execute_reply":"2023-06-12T18:33:55.298731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport librosa.display\nimport numpy as np\nfrom PIL import Image\nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib.cm as cm\nimport csv\nfrom operator import length_hint\nimport glob","metadata":{"execution":{"iopub.status.busy":"2023-06-12T18:33:55.302135Z","iopub.execute_input":"2023-06-12T18:33:55.303049Z","iopub.status.idle":"2023-06-12T18:33:55.350204Z","shell.execute_reply.started":"2023-06-12T18:33:55.303006Z","shell.execute_reply":"2023-06-12T18:33:55.348697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def create_melspectrogram(audio_path, start_time, end_time, save_path):\n#     # Load the audio file\n#     audio, sr = librosa.load(audio_path, offset=start_time, duration=(end_time - start_time))\n\n#     # Compute the Mel spectrogram\n#     mel_spectrogram = librosa.feature.melspectrogram(y=audio, sr=sr)\n#     mel_spectrogram_db = librosa.power_to_db(mel_spectrogram, ref=np.max)\n\n#     # Normalize the spectrogram to the 0-255 range\n#     mel_spectrogram_normalized = (mel_spectrogram_db - mel_spectrogram_db.min()) / (mel_spectrogram_db.max() - mel_spectrogram_db.min()) * 255\n#     mel_spectrogram_normalized = mel_spectrogram_normalized.astype(np.uint8)\n\n#     # Apply a colormap\n#     colormap = cm.get_cmap('viridis')  # Choose a colormap (e.g., 'viridis', 'inferno', 'plasma', etc.)\n#     mel_spectrogram_colored = colormap(mel_spectrogram_normalized)\n\n#     # Convert the spectrogram to a PIL image\n#     image = Image.fromarray((mel_spectrogram_colored[:, :, :3] * 255).astype(np.uint8))  # Keep only RGB channels\n\n#     # Save the image as a PNG file\n#     image.save(save_path + '.png', format='PNG')\n#     # Save the array\n#     mel_spec_np = np.array(mel_spectrogram)\n#     np.save(save_path + '.npy', mel_spec_np)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T18:33:55.351891Z","iopub.execute_input":"2023-06-12T18:33:55.352477Z","iopub.status.idle":"2023-06-12T18:33:55.358848Z","shell.execute_reply.started":"2023-06-12T18:33:55.352438Z","shell.execute_reply":"2023-06-12T18:33:55.357522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_melspectrogram(audio_path, start_time, end_time, save_path):\n    # Load the audio file\n    audio, sr = librosa.load(audio_path, offset=start_time, duration=(end_time - start_time))\n\n    # Compute the Mel spectrogram\n    mel_spectrogram = librosa.feature.melspectrogram(y=audio, sr=sr)\n    mel_spectrogram_db = librosa.power_to_db(mel_spectrogram, ref=np.max)\n\n    # Normalize the spectrogram to the 0-255 range\n    mel_spectrogram_normalized = (mel_spectrogram_db - mel_spectrogram_db.min()) / (mel_spectrogram_db.max() - mel_spectrogram_db.min()) * 255\n    mel_spectrogram_normalized = mel_spectrogram_normalized.astype(np.uint8)\n\n    # Apply a colormap\n    colormap = cm.get_cmap('viridis')  # Choose a colormap (e.g., 'viridis', 'inferno', 'plasma', etc.)\n    mel_spectrogram_colored = colormap(mel_spectrogram_normalized)\n\n    # Display the spectrogram with axis labels\n    plt.figure(figsize=(10, 4))\n    librosa.display.specshow(mel_spectrogram_db, sr=sr, x_axis='time', y_axis='mel')\n    plt.colorbar(format='%+2.0f dB')  # Display colorbar with dB scale\n    plt.xlabel('Time')\n    plt.ylabel('Mel Frequency')\n    plt.title('Mel Spectrogram')\n    plt.tight_layout()\n\n    # Save the plot as a PNG file\n    plt.savefig(save_path + '_plot.png', format='PNG')\n    plt.close()  # Close the plot to free up resources\n\n    # Convert the spectrogram to a PIL image\n    image = Image.fromarray((mel_spectrogram_colored[:, :, :3] * 255).astype(np.uint8))  # Keep only RGB channels\n\n    # Save the PIL image as a PNG file\n    image.save(save_path + '_image.png', format='PNG')\n\n    # Save the array\n    mel_spec_np = np.array(mel_spectrogram)\n    np.save(save_path + '.npy', mel_spec_np)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T18:33:55.361985Z","iopub.execute_input":"2023-06-12T18:33:55.362316Z","iopub.status.idle":"2023-06-12T18:33:55.375301Z","shell.execute_reply.started":"2023-06-12T18:33:55.362289Z","shell.execute_reply":"2023-06-12T18:33:55.374192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image(path):\n    Image.open(path)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T18:33:55.376770Z","iopub.execute_input":"2023-06-12T18:33:55.377148Z","iopub.status.idle":"2023-06-12T18:33:55.392537Z","shell.execute_reply.started":"2023-06-12T18:33:55.377117Z","shell.execute_reply":"2023-06-12T18:33:55.391203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_array(path):\n    loaded_mel_spec_np = np.load(path)\n    return loaded_mel_spec_np","metadata":{"execution":{"iopub.status.busy":"2023-06-12T18:33:55.394065Z","iopub.execute_input":"2023-06-12T18:33:55.394616Z","iopub.status.idle":"2023-06-12T18:33:55.404127Z","shell.execute_reply.started":"2023-06-12T18:33:55.394579Z","shell.execute_reply":"2023-06-12T18:33:55.403041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_soundscape_label_df = pd.read_csv('/kaggle/input/birdclef-2021/train_soundscape_labels.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-06-12T18:33:55.405752Z","iopub.execute_input":"2023-06-12T18:33:55.406080Z","iopub.status.idle":"2023-06-12T18:33:55.416356Z","shell.execute_reply.started":"2023-06-12T18:33:55.406054Z","shell.execute_reply":"2023-06-12T18:33:55.415179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"soundscapeCSV = '/kaggle/input/birdclef-2021/train_soundscape_labels.csv'\nwith open(soundscapeCSV, 'r') as inputFile:\n    soundscapeLabels = []\n    csvReader = csv.DictReader(inputFile)\n    for row in csvReader:\n        soundscapeLabels.append(dict(row))\n\n# print(soundscapeLabels)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T18:33:55.418194Z","iopub.execute_input":"2023-06-12T18:33:55.418938Z","iopub.status.idle":"2023-06-12T18:33:55.446391Z","shell.execute_reply.started":"2023-06-12T18:33:55.418907Z","shell.execute_reply":"2023-06-12T18:33:55.445430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example usage\nsoundscapesMelsDir = 'test_soundscapes_mels'\nsoundscapesDir = '/kaggle/input/birdclef-2021/train_soundscapes'\n\n# os.makedirs(soundscapesMelsDir)\n\nfor row in soundscapeLabels:\n    prefix = row['audio_id']\n    audio_path = glob.glob(soundscapesDir + '/' + prefix + \"*\")[0]\n\n    end_time = int(row['seconds']) \n    if (end_time > 60):\n        continue\n    start_time = end_time - 5\n    if (not os.path.exists(soundscapesMelsDir + '/' + prefix)):\n        os.makedirs(soundscapesMelsDir + '/' + prefix)\n    save_path = soundscapesMelsDir + '/' + prefix + '/' + row['row_id'] + '_' + row['birds']\n    create_melspectrogram(audio_path, start_time, end_time, save_path)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T18:33:55.448443Z","iopub.execute_input":"2023-06-12T18:33:55.448862Z","iopub.status.idle":"2023-06-12T18:36:18.356355Z","shell.execute_reply.started":"2023-06-12T18:33:55.448824Z","shell.execute_reply":"2023-06-12T18:36:18.355472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_id = '44957'\nnum = 8\nsec = 5 * num\ndir = '/kaggle/working/' + soundscapesMelsDir\n\nname = audio_id + '_' + str(sec)\n\n# image_path = dir + '/' + audio_id + '/' + name + '.png'\n# array_path = dir + '/' + audio_id + '/' + name + '.npy'\n\nimage_path = glob.glob(dir + '/' + audio_id + '/' + audio_id + '_*_' + str(sec) + '_*_plot.png')[0]\nimage_path2 = glob.glob(dir + '/' + audio_id + '/' + audio_id + '_*_' + str(sec) + '_*_image.png')[0]\n\n# display_image(image_path)\nprint(image_path)\nImage.open(image_path)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-12T18:46:16.438981Z","iopub.execute_input":"2023-06-12T18:46:16.439392Z","iopub.status.idle":"2023-06-12T18:46:16.514131Z","shell.execute_reply.started":"2023-06-12T18:46:16.439361Z","shell.execute_reply":"2023-06-12T18:46:16.513041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Image.open(image_path2)","metadata":{"execution":{"iopub.status.busy":"2023-06-12T18:44:37.041087Z","iopub.execute_input":"2023-06-12T18:44:37.041460Z","iopub.status.idle":"2023-06-12T18:44:37.064827Z","shell.execute_reply.started":"2023-06-12T18:44:37.041432Z","shell.execute_reply":"2023-06-12T18:44:37.063722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_db_value_at_time_freq(loaded_mel_spec_np, f, t, sr, hop_length, n_fft):\n    # Calculate frame index\n    frame_index = int(t * sr / hop_length)\n\n    # Convert mel frequency to frequency bin\n    mel_frequencies = librosa.mel_frequencies(n_mels=loaded_mel_spec_np.shape[0], fmin=0, fmax=sr/2)\n    f_bin = np.argmin(np.abs(mel_frequencies - f))\n\n    # Check if frame index or frequency bin is out of bounds\n    if frame_index >= loaded_mel_spec_np.shape[1]:\n        raise ValueError(\"Frame index is out of bounds\")\n    if f_bin >= loaded_mel_spec_np.shape[0]:\n        raise ValueError(\"Frequency bin is out of bounds\")\n\n    # Retrieve dB value\n    mel_spectrogram_db = librosa.power_to_db(loaded_mel_spec_np, ref=np.max)\n    db_value = mel_spectrogram_db[f_bin, frame_index]\n\n    return db_value\n","metadata":{"execution":{"iopub.status.busy":"2023-06-12T18:36:18.477609Z","iopub.execute_input":"2023-06-12T18:36:18.478300Z","iopub.status.idle":"2023-06-12T18:36:18.487882Z","shell.execute_reply.started":"2023-06-12T18:36:18.478261Z","shell.execute_reply":"2023-06-12T18:36:18.486481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"t = 3.3  # time index\nf = 3072  # mel frequency \narray_path = glob.glob(dir + '/' + audio_id + '/' + audio_id + '*' + str(sec) + '*' + '.npy')[0]\n\nsoundscape_arr = get_array(array_path)\n\n\nres = get_db_value_at_time_freq(soundscape_arr, f, t, 32_000, 512, 2048)\nres","metadata":{"execution":{"iopub.status.busy":"2023-06-12T18:46:19.934794Z","iopub.execute_input":"2023-06-12T18:46:19.935901Z","iopub.status.idle":"2023-06-12T18:46:19.947716Z","shell.execute_reply.started":"2023-06-12T18:46:19.935842Z","shell.execute_reply":"2023-06-12T18:46:19.946392Z"},"trusted":true},"execution_count":null,"outputs":[]}]}