{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# What's this notebook about?\n\nIn this notebook I will try to blend code this [public notebook](https://www.kaggle.com/code/ashokkumarbibbab/birds-classification-in-rfc-model-accuracy-87/notebook) and this [github repo](https://github.com/marathomas/tutorial_repo) to explore dimensionally reduced representation of bird's calls.","metadata":{}},{"cell_type":"markdown","source":"## Install and import libraries","metadata":{}},{"cell_type":"code","source":"#libraries for file operations\nimport os\nimport glob\nimport shutil #High-level file operations\nimport zipfile\nimport pickle\nfrom joblib import dump, load\nfrom pathlib import Path\n\n#plotting\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_theme()\n!pip install plotly -q\nimport plotly.express as px\n\n#audio analisys\nimport librosa # package for music and audio analysis\nfrom IPython.display import Audio\n\n#DataFrame and arrays operations\nimport numpy as np\nimport pandas as pd\n\n#ML libs\n!pip install -U imbalanced-learn\nfrom imblearn.over_sampling import RandomOverSampler\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, confusion_matrix\n\nimport umap\n\nfrom tqdm import tqdm\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-09T13:41:04.035291Z","iopub.execute_input":"2024-04-09T13:41:04.036586Z","iopub.status.idle":"2024-04-09T13:42:11.467263Z","shell.execute_reply.started":"2024-04-09T13:41:04.036507Z","shell.execute_reply":"2024-04-09T13:42:11.465502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Collection and Processing","metadata":{}},{"cell_type":"markdown","source":"### Read metadata table","metadata":{}},{"cell_type":"code","source":"meta_data = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\nmeta_data.head(4)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:58:26.785127Z","iopub.execute_input":"2024-04-09T12:58:26.786060Z","iopub.status.idle":"2024-04-09T12:58:27.026862Z","shell.execute_reply.started":"2024-04-09T12:58:26.786016Z","shell.execute_reply":"2024-04-09T12:58:27.025606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_data.info()","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:58:27.028528Z","iopub.execute_input":"2024-04-09T12:58:27.028904Z","iopub.status.idle":"2024-04-09T12:58:27.085524Z","shell.execute_reply.started":"2024-04-09T12:58:27.028874Z","shell.execute_reply":"2024-04-09T12:58:27.084195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Not all rows contain `latitude` and `longitude` data.","metadata":{}},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"code","source":"fig = px.scatter_mapbox(meta_data, lat='latitude', lon='longitude', color='common_name', \n                        hover_name='common_name', hover_data=['latitude', 'longitude'], \n                        title='Origin of Bird Species',\n                        zoom=1, height=600)\nfig.update_layout(\n    mapbox_style=\"white-bg\",\n    mapbox_layers=[\n        {\n            \"below\": 'traces',\n            \"sourcetype\": \"raster\",\n            \"sourceattribution\": \"United States Geological Survey\",\n            \"source\": [\n                \"https://basemap.nationalmap.gov/arcgis/rest/services/USGSImageryOnly/MapServer/tile/{z}/{y}/{x}\"\n            ]\n        }\n      ])\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:58:27.089408Z","iopub.execute_input":"2024-04-09T12:58:27.089858Z","iopub.status.idle":"2024-04-09T12:58:30.360853Z","shell.execute_reply.started":"2024-04-09T12:58:27.089824Z","shell.execute_reply":"2024-04-09T12:58:30.358388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most data is collected in Europe.","metadata":{}},{"cell_type":"code","source":"def audio_waveframe(file_path):\n    # Load the audio file\n    audio_data, sampling_rate = librosa.load(file_path)\n    # Calculate the duration of the audio file\n    duration = len(audio_data) / sampling_rate\n    # Create a time array for plotting\n    time = np.arange(0, duration, 1/sampling_rate)\n    # Plot the waveform\n    plt.figure(figsize=(30, 4))\n    plt.plot(time, audio_data, color='blue')\n    plt.title('Audio Waveform')\n    plt.xlabel('Time (s)')\n    plt.ylabel('Amplitude')\n    plot = plt.show()\n    return plot\n\ndef spectrogram(file_path):\n    # Compute the short-time Fourier transform (STFT)\n    n_fft = 500  # Number of FFT points 2048\n    hop_length = 50  # Hop length for STFT 512\n    audio_data, sampling_rate = librosa.load(file_path)\n    stft = librosa.stft(audio_data, n_fft=n_fft, hop_length=hop_length)\n    # Convert the magnitude spectrogram to decibels (log scale)\n    spectrogram = librosa.amplitude_to_db(np.abs(stft))\n    # Plot the spectrogram\n    plt.figure(figsize=(30, 6))\n    librosa.display.specshow(spectrogram, sr=sampling_rate, hop_length=hop_length, x_axis='time', y_axis='linear')\n    plt.colorbar(format='%+2.0f dB')\n    plt.title('Spectrogram')\n    plt.xlabel('Time (s)')\n    plt.ylabel('Frequency (Hz)')\n    plt.tight_layout()\n    plot = plt.show()\n    return plot\n\ndef audio_analysis(file_path):\n    aw = audio_waveframe(file_path)\n    spg = spectrogram(file_path)\n    return aw, spg","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:58:30.362478Z","iopub.execute_input":"2024-04-09T12:58:30.363673Z","iopub.status.idle":"2024-04-09T12:58:30.379517Z","shell.execute_reply.started":"2024-04-09T12:58:30.363633Z","shell.execute_reply":"2024-04-09T12:58:30.374925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_analysis('/kaggle/input/birdclef-2024/train_audio/asbfly/XC134896.ogg')\nAudio('/kaggle/input/birdclef-2024/train_audio/asbfly/XC134896.ogg')","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:58:30.381636Z","iopub.execute_input":"2024-04-09T12:58:30.382114Z","iopub.status.idle":"2024-04-09T12:58:49.636058Z","shell.execute_reply.started":"2024-04-09T12:58:30.382069Z","shell.execute_reply":"2024-04-09T12:58:49.634746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_analysis('/kaggle/input/birdclef-2024/train_audio/asbfly/XC164848.ogg')\nAudio('/kaggle/input/birdclef-2024/train_audio/asbfly/XC164848.ogg')","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:58:49.637951Z","iopub.execute_input":"2024-04-09T12:58:49.638661Z","iopub.status.idle":"2024-04-09T12:58:52.724404Z","shell.execute_reply.started":"2024-04-09T12:58:49.638625Z","shell.execute_reply":"2024-04-09T12:58:52.723057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Audio from one bird species can have big differences from each other.\n* Noise in the background can couse troubles when building models.","metadata":{}},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"code","source":"# Path to the directory containing your audio dataset\ndataset_dir = '/kaggle/input/birdclef-2024/train_audio'\n# Initialize an empty dictionary to store the mapping between audio files and labels\nlabel_mapping = {}\n# Iterate over subdirectories (classes) in the dataset directory\nfor label in os.listdir(dataset_dir):\n    label_dir = os.path.join(dataset_dir, label)\n    # Check if the item in the dataset directory is a directory\n    if os.path.isdir(label_dir):\n        # Iterate over audio files in the subdirectory (class)\n        for audio_file in os.listdir(label_dir):\n            # Add the mapping between audio file path and label to the dictionary\n            audio_file_path = os.path.join(label_dir, audio_file)\n            label_mapping[audio_file_path] = label","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:58:52.725957Z","iopub.execute_input":"2024-04-09T12:58:52.726337Z","iopub.status.idle":"2024-04-09T12:58:55.959889Z","shell.execute_reply.started":"2024-04-09T12:58:52.726304Z","shell.execute_reply":"2024-04-09T12:58:55.958797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a list of tuples containing the audio file paths and labels\ndata = [(audio_file_path, label) for audio_file_path, label in label_mapping.items()]\n# Create a Pandas DataFrame from the list of tuples\nannotated_data = pd.DataFrame(data, columns=['audio_file_path', 'label'])\nannotated_data.sample(5)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:58:55.961769Z","iopub.execute_input":"2024-04-09T12:58:55.962548Z","iopub.status.idle":"2024-04-09T12:58:55.994075Z","shell.execute_reply.started":"2024-04-09T12:58:55.962503Z","shell.execute_reply":"2024-04-09T12:58:55.993062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's see distribution of audio lenths","metadata":{}},{"cell_type":"code","source":"tqdm.pandas()\nannotated_data['duration'] = annotated_data['audio_file_path'].progress_apply(lambda row: librosa.get_duration(path=row))\n\nplt.title('Distribution of audio durations')\nsns.histplot(data = annotated_data, x='duration', log_scale=True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-09T13:16:21.153177Z","iopub.execute_input":"2024-04-09T13:16:21.154585Z","iopub.status.idle":"2024-04-09T13:21:23.253126Z","shell.execute_reply.started":"2024-04-09T13:16:21.154545Z","shell.execute_reply":"2024-04-09T13:21:23.251860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"longest_data = annotated_data.loc[annotated_data['duration']==annotated_data['duration'].max(), 'audio_file_path'].values[0]\nshortest_data = annotated_data.loc[annotated_data['duration']==annotated_data['duration'].min(), 'audio_file_path'].values[0]","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:59:32.341417Z","iopub.status.idle":"2024-04-09T12:59:32.341935Z","shell.execute_reply.started":"2024-04-09T12:59:32.341704Z","shell.execute_reply":"2024-04-09T12:59:32.341724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Shortest recording duration: {annotated_data.loc[annotated_data['duration']==annotated_data['duration'].min(), 'duration']}\")\nprint()\nprint(f\"Longest recording duration: {annotated_data.loc[annotated_data['duration']==annotated_data['duration'].max(), 'duration']}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:59:32.343741Z","iopub.status.idle":"2024-04-09T12:59:32.344421Z","shell.execute_reply.started":"2024-04-09T12:59:32.344092Z","shell.execute_reply":"2024-04-09T12:59:32.344120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Duration difference is realy big. It begins with half a second and ends with more than 16 hours. I think, we can trim audio lenth to 1 minut for more efficient computing.","metadata":{}},{"cell_type":"code","source":"print('Longest data')\naudio_waveframe(longest_data)\n# audio_analysis(longest_data) Requires too much memory\n# Audio(longest_data)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:59:32.346325Z","iopub.status.idle":"2024-04-09T12:59:32.346926Z","shell.execute_reply.started":"2024-04-09T12:59:32.346685Z","shell.execute_reply":"2024-04-09T12:59:32.346706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Shortest data')\naudio_analysis(shortest_data)\nAudio(shortest_data)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:59:32.351865Z","iopub.status.idle":"2024-04-09T12:59:32.353101Z","shell.execute_reply.started":"2024-04-09T12:59:32.352797Z","shell.execute_reply":"2024-04-09T12:59:32.352827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature Extraction\n\n","metadata":{}},{"cell_type":"markdown","source":"Here I will try to remove background noise from bird's vocals. Approach below seems to work well in most cases. But sometimes it can keep the microphone noise and drop bird's vocal.","metadata":{}},{"cell_type":"markdown","source":"Read audio file","metadata":{}},{"cell_type":"markdown","source":"Plot spectrum","metadata":{}},{"cell_type":"markdown","source":"### Background removal by nearest neighbours method","metadata":{}},{"cell_type":"code","source":"def background_removal_test(path):\n    # read file\n    audio, sample_rate = librosa.load(path)\n    \n    # set index to to get first 60 seconds of data\n    idx = slice(*librosa.time_to_frames([0, 60], sr=sample_rate))\n    \n    # keep only 1 min of audio\n    audio = audio[:60*sample_rate]\n\n    # And compute the spectrogram magnitude and phase\n    S_full, phase = librosa.magphase(librosa.stft(audio))\n\n    # We'll compare frames using cosine similarity, and aggregate similar frames\n    # by taking their (per-frequency) median value.\n    #\n    # To avoid being biased by local continuity, we constrain similar frames to be\n    # separated by at least 2 seconds.\n    #\n    # This suppresses sparse/non-repetetitive deviations from the average spectrum,\n    # and works well to discard vocal elements.\n\n    S_filter = librosa.decompose.nn_filter(S_full,\n                                           aggregate=np.median,\n                                           metric='cosine',\n                                           width=int(librosa.time_to_frames(2, sr=sample_rate)))\n\n    # The output of the filter shouldn't be greater than the input\n    # if we assume signals are additive.  Taking the pointwise minimium\n    # with the input spectrum forces this.\n    S_filter = np.minimum(S_full, S_filter)\n    \n    # We can also use a margin to reduce bleed between the vocals and instrumentation masks.\n    # Note: the margins need not be equal for foreground and background separation\n    margin_i, margin_v = 2, 5\n    power = 2\n\n    mask_i = librosa.util.softmask(S_filter,\n                                   margin_i * (S_full - S_filter),\n                                   power=power)\n\n    mask_v = librosa.util.softmask(S_full - S_filter,\n                                   margin_v * S_filter,\n                                   power=power)\n\n    # Once we have the masks, simply multiply them with the input spectrum\n    # to separate the components\n\n    S_foreground = mask_v * S_full\n    S_background = mask_i * S_full\n    \n    plt.figure(figsize=(12, 8))\n    plt.subplot(3, 1, 1)\n    librosa.display.specshow(librosa.amplitude_to_db(S_full[:, idx], ref=np.max),\n                             y_axis='log', sr=sample_rate)\n    plt.title('Full spectrum')\n    plt.colorbar()\n\n    plt.subplot(3, 1, 2)\n    librosa.display.specshow(librosa.amplitude_to_db(S_background[:, idx], ref=np.max),\n                             y_axis='log', sr=sample_rate)\n    plt.title('Background')\n    plt.colorbar()\n    plt.subplot(3, 1, 3)\n    librosa.display.specshow(librosa.amplitude_to_db(S_foreground[:, idx], ref=np.max),\n                             y_axis='log', x_axis='time', sr=sample_rate)\n    plt.title('Foreground')\n    plt.colorbar()\n    plt.tight_layout()\n    plt.show()\n    \n    return","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:59:51.279580Z","iopub.execute_input":"2024-04-09T12:59:51.280058Z","iopub.status.idle":"2024-04-09T12:59:51.297676Z","shell.execute_reply.started":"2024-04-09T12:59:51.280023Z","shell.execute_reply":"2024-04-09T12:59:51.296099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Example of good noise removal\")\nbackground_removal_test('/kaggle/input/birdclef-2024/train_audio/asbfly/XC134896.ogg')","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:59:52.365214Z","iopub.execute_input":"2024-04-09T12:59:52.365780Z","iopub.status.idle":"2024-04-09T13:00:00.359425Z","shell.execute_reply.started":"2024-04-09T12:59:52.365739Z","shell.execute_reply":"2024-04-09T13:00:00.357870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Example of bad noise removal\")\nbackground_removal_test('/kaggle/input/birdclef-2024/train_audio/comkin1/XC267325.ogg')","metadata":{"execution":{"iopub.status.busy":"2024-04-09T13:00:00.361830Z","iopub.execute_input":"2024-04-09T13:00:00.362249Z","iopub.status.idle":"2024-04-09T13:00:04.485738Z","shell.execute_reply.started":"2024-04-09T13:00:00.362215Z","shell.execute_reply":"2024-04-09T13:00:04.484342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Background removal by median","metadata":{}},{"cell_type":"code","source":"def median_removal_test(path):\n    # read file\n    audio, sample_rate = librosa.load(path)\n    \n    # set index to to get first 60 seconds of data\n    idx = slice(*librosa.time_to_frames([0, 60], sr=sample_rate))\n    \n    # keep only 1 min of audio\n    audio = audio[:60*sample_rate]\n\n    # And compute the spectrogram magnitude and phase\n    S_full, phase = librosa.magphase(librosa.stft(audio))\n\n\n    median_substracted_audio = S_full - np.median(S_full)\n    \n    plt.figure(figsize=(12, 8))\n    plt.subplot(3, 1, 1)\n    librosa.display.specshow(librosa.amplitude_to_db(S_full[:, idx], ref=np.max),\n                             y_axis='log', sr=sample_rate)\n    plt.title('Full spectrum')\n\n    plt.subplot(3, 1, 2)\n    librosa.display.specshow(librosa.amplitude_to_db(median_substracted_audio[:, idx], ref=np.max),\n                             y_axis='log', x_axis='time', sr=sample_rate)\n    plt.title('Denoised')\n    plt.colorbar()\n    plt.tight_layout()\n    plt.show()\n    \n    return","metadata":{"execution":{"iopub.status.busy":"2024-04-09T13:10:50.238736Z","iopub.execute_input":"2024-04-09T13:10:50.239243Z","iopub.status.idle":"2024-04-09T13:10:50.251812Z","shell.execute_reply.started":"2024-04-09T13:10:50.239209Z","shell.execute_reply":"2024-04-09T13:10:50.249942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Example of bad noise removal\")\nmedian_removal_test('/kaggle/input/birdclef-2024/train_audio/asbfly/XC134896.ogg')","metadata":{"execution":{"iopub.status.busy":"2024-04-09T13:10:50.696696Z","iopub.execute_input":"2024-04-09T13:10:50.697239Z","iopub.status.idle":"2024-04-09T13:10:53.419632Z","shell.execute_reply.started":"2024-04-09T13:10:50.697200Z","shell.execute_reply":"2024-04-09T13:10:53.418355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Example of bad noise removal\")\nmedian_removal_test('/kaggle/input/birdclef-2024/train_audio/comkin1/XC267325.ogg')","metadata":{"execution":{"iopub.status.busy":"2024-04-09T13:10:53.422005Z","iopub.execute_input":"2024-04-09T13:10:53.422498Z","iopub.status.idle":"2024-04-09T13:10:55.045921Z","shell.execute_reply.started":"2024-04-09T13:10:53.422458Z","shell.execute_reply":"2024-04-09T13:10:55.044473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Substraction of the median values somehow works ","metadata":{}},{"cell_type":"code","source":"# Function to extract features from audio file\ndef extract_features(file_path, margin_v=2, power=2):\n    # Load and cut audio file\n    audio, sample_rate = librosa.load(file_path)\n    audio = audio[:60*sample_rate]\n    \n    \n    # deniose by substarcting median\n    audio -= np.median(audio, axis=0)\n    \n    '''\n       This code takes too long to compute\n       Replace it with median substraction\n    '''\n#     # Compute the spectrogram magnitude and phase\n#     S_full, phase = librosa.magphase(librosa.stft(audio))\n    \n    \n#     # Filter noise\n#     S_filter = librosa.decompose.nn_filter(S_full,\n#                                            aggregate=np.median,\n#                                            metric='cosine',\n#                                            width=int(librosa.time_to_frames(2, sr=sample_rate)))\n\n#     S_filter = np.minimum(S_full, S_filter)\n\n#     mask_v = librosa.util.softmask(S_full - S_filter,\n#                                    margin_v * S_filter,\n#                                    power=power)\n\n#     S_foreground = mask_v * S_full\n\n#     # Extract features using Mel-Frequency Cepstral Coefficients (MFCC)\n#     mfccs = librosa.feature.mfcc(y=S_foreground, sr=sample_rate, n_mfcc=40)\n    \n    mfccs = librosa.feature.mfcc(y=audio, sr=sample_rate, n_mfcc=40)\n    # Flatten the features into a 1D array\n    flattened_features = np.mean(mfccs.T, axis=0)\n    return flattened_features\n\n# Function to load dataset and extract features\ndef load_data_and_extract_features(data_dir):\n    labels = []\n    features = []\n    # Loop through each audio file in the dataset directory\n    for filename in os.listdir(data_dir):\n        if filename.endswith('.ogg'):\n            file_path = os.path.join(data_dir, filename)\n            # Extract label from filename\n            label = filename.split('-')[0]\n            labels.append(label)\n            # Extract features from audio file\n            feature = extract_features(file_path)\n            features.append(feature)\n    return np.array(features), np.array(labels)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T13:12:42.698579Z","iopub.execute_input":"2024-04-09T13:12:42.698992Z","iopub.status.idle":"2024-04-09T13:12:42.709428Z","shell.execute_reply.started":"2024-04-09T13:12:42.698964Z","shell.execute_reply":"2024-04-09T13:12:42.708502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"`extract_features(file_path)`:\n\n* Loads an audio file using librosa library.\n\n* Some additions to [ASHOK KUMARASHOK KUMAR's](https://www.kaggle.com/code/ashokkumarbibbab/birds-classification-in-rfc-model-accuracy-87/notebook) feature extraction. In this step we will separate vocals from background.\n\n* Calculates Mel-Frequency Cepstral Coefficients (MFCC) features from the audio.\n\n* Averages the MFCC features over time to create a flattened feature vector.\n\n* Returns the flattened feature vector.\n\n\n`load_data_and_extract_features(data_dir)`:\n\n* Iterates through each audio file in the specified directory.\n\n* Extracts the label from the filename by splitting it at '-'.\n\n* Calls extract_features() to extract features from each audio file.\n\n* Returns numpy arrays containing the extracted features and corresponding labels.extract_features(file_path):","metadata":{}},{"cell_type":"code","source":"extracted_features = []\n\nfor i in tqdm(annotated_data['audio_file_path']):\n    features = extract_features(file_path=i)\n    extracted_features.append(features)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T13:30:26.016753Z","iopub.execute_input":"2024-04-09T13:30:26.018179Z","iopub.status.idle":"2024-04-09T13:30:38.329290Z","shell.execute_reply.started":"2024-04-09T13:30:26.018118Z","shell.execute_reply":"2024-04-09T13:30:38.327428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"extracted_features\", \"wb\") as file:   #Pickling\n    pickle.dump(extracted_features, file)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T13:16:09.384729Z","iopub.status.idle":"2024-04-09T13:16:09.385316Z","shell.execute_reply.started":"2024-04-09T13:16:09.385020Z","shell.execute_reply":"2024-04-09T13:16:09.385048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with open(\"/kaggle/input/extracted-features-pickle/extracted_features\", \"rb\") as file:   # Unpickling\n#     pickled_extracted_features = pickle.load(file)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T12:59:32.369459Z","iopub.status.idle":"2024-04-09T12:59:32.369904Z","shell.execute_reply.started":"2024-04-09T12:59:32.369699Z","shell.execute_reply":"2024-04-09T12:59:32.369717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let's try to create UMAP plot","metadata":{}},{"cell_type":"markdown","source":"Prepare feautres for analysis.","metadata":{}},{"cell_type":"code","source":"# z_transformer = StandardScaler() # z-transform each spectrogram separatly\n# specs = [z_transformer.fit_transform(features.reshape(-1, 1)) for features in extracted_features]\n\nspecs = extracted_features\nflattened_specs = [spec.flatten() for spec in specs] # pad all specs to maxlen, then row-wise concatenate (flatten)\ndata = np.asarray(flattened_specs) # data is the final input data for UMAP","metadata":{"execution":{"iopub.status.busy":"2024-04-09T13:59:56.075486Z","iopub.execute_input":"2024-04-09T13:59:56.076560Z","iopub.status.idle":"2024-04-09T13:59:56.086168Z","shell.execute_reply.started":"2024-04-09T13:59:56.076506Z","shell.execute_reply":"2024-04-09T13:59:56.084235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Specify UMAP parameters ","metadata":{}},{"cell_type":"code","source":"METRIC_TYPE = 'euclidean'     # distance metric used in UMAP. Check UMAP documentation for other options\n                              # e.g. 'euclidean', correlation', 'cosine','manhattan' ...\n    \nN_COMP = 3                    # number of dimensions desired in latent space  \n\nreducer = umap.UMAP(n_components=N_COMP, metric = METRIC_TYPE,  # specify parameters of UMAP reducer\n                    min_dist = 0, random_state=2204) ","metadata":{"execution":{"iopub.status.busy":"2024-04-09T14:00:00.760124Z","iopub.execute_input":"2024-04-09T14:00:00.761239Z","iopub.status.idle":"2024-04-09T14:00:00.768781Z","shell.execute_reply.started":"2024-04-09T14:00:00.761178Z","shell.execute_reply":"2024-04-09T14:00:00.767300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Calculate dimensionally reduced representation","metadata":{}},{"cell_type":"code","source":"embedding = reducer.fit_transform(data)","metadata":{"execution":{"iopub.status.busy":"2024-04-09T14:00:02.586100Z","iopub.execute_input":"2024-04-09T14:00:02.586638Z","iopub.status.idle":"2024-04-09T14:00:04.024597Z","shell.execute_reply.started":"2024-04-09T14:00:02.586601Z","shell.execute_reply":"2024-04-09T14:00:04.022675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"annotated_data[\"UMAP_x\"] = embedding[:,0]\nannotated_data[\"UMAP_y\"] = embedding[:,1]\nannotated_data[\"UMAP_z\"] = embedding[:,2]","metadata":{"execution":{"iopub.status.busy":"2024-04-09T14:00:04.954993Z","iopub.execute_input":"2024-04-09T14:00:04.955527Z","iopub.status.idle":"2024-04-09T14:00:04.964115Z","shell.execute_reply.started":"2024-04-09T14:00:04.955488Z","shell.execute_reply":"2024-04-09T14:00:04.962580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# px.scatter_3d(x=annotated_data[\"UMAP_x\"], y=annotated_data[\"UMAP_y\"], \n#            z=annotated_data[\"UMAP_z\"],\n#            color=annotated_data['label'])","metadata":{"execution":{"iopub.status.busy":"2024-04-09T14:00:33.592423Z","iopub.execute_input":"2024-04-09T14:00:33.592966Z","iopub.status.idle":"2024-04-09T14:00:33.598974Z","shell.execute_reply.started":"2024-04-09T14:00:33.592926Z","shell.execute_reply":"2024-04-09T14:00:33.597644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.scatter(x=annotated_data[\"UMAP_x\"], y=annotated_data[\"UMAP_y\"], \n           color=annotated_data['label'])","metadata":{"execution":{"iopub.status.busy":"2024-04-09T14:00:09.930995Z","iopub.execute_input":"2024-04-09T14:00:09.931547Z","iopub.status.idle":"2024-04-09T14:00:10.036280Z","shell.execute_reply.started":"2024-04-09T14:00:09.931505Z","shell.execute_reply":"2024-04-09T14:00:10.034673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}