{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":2301966,"sourceType":"datasetVersion","datasetId":1388111},{"sourceId":181633964,"sourceType":"kernelVersion"}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Related Work \n\n- Extensive Data Exploration: https://www.kaggle.com/code/virajkadam/birdclef-2024-extenstive-eda\n\n- Extracting Bird features using PHI-3.8-mini: https://www.kaggle.com/code/virajkadam/birdclef2024-extract-features-using-phi-3\n\n- Training a Multimodal text to embedding model : https://www.kaggle.com/code/virajkadam/birdclef-training-a-multimodal-text-image-model","metadata":{}},{"cell_type":"markdown","source":"# Imports ","metadata":{}},{"cell_type":"code","source":"\n!pip install audiomentations -q\n!pip install pqdm -q\n\nimport os\nimport pandas as pd\nimport numpy as np\nimport json\n\nfrom pathlib import Path\nimport pqdm\nfrom tqdm import tqdm\n\n\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\n\nimport warnings\nwarnings.filterwarnings(action='ignore')\n\n#audio\nimport librosa\nfrom IPython.display import Audio\n#audio augmentations'\nfrom audiomentations import Compose,AddGaussianSNR,Shift,TimeStretch,TimeMask,PolarityInversion","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:29:47.759827Z","iopub.execute_input":"2024-06-09T14:29:47.761035Z","iopub.status.idle":"2024-06-09T14:30:18.980463Z","shell.execute_reply.started":"2024-06-09T14:29:47.760992Z","shell.execute_reply":"2024-06-09T14:30:18.978717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    RANDOM_SEED = 7\n    SAMPLE_RATE = 32000\n    N_ffts = 1024\n    Window_size = 1024\n    SIGNAL_LENGTH = 5 # seconds\n    FMIN = 1000\n    FMAX = 10000\n    hop_length = 320\n    n_mels = 224\n","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:30:18.984012Z","iopub.execute_input":"2024-06-09T14:30:18.984533Z","iopub.status.idle":"2024-06-09T14:30:18.992143Z","shell.execute_reply.started":"2024-06-09T14:30:18.984484Z","shell.execute_reply":"2024-06-09T14:30:18.990656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extra data to filter no calls : Freefield Data\n\n    The longer audios might have a lot of empty calls , where there are no bird calls. to filter these examples when training, we can take a few examples of no bird calls, and use them in the final training\n\n    This dataset contains 7690 10-second audio files in a standardised format, extracted from contributions on the Freesound archive which were labelled with the \"field-recording\" tag. Note that the original tagging (as well as the audio submission) is crowdsourced, so the dataset is not guaranteed to consist purely of \"field recordings\" as might be defined by practitioners. The intention is to represent the content of an archive collection on such a topic, rather than to represent a controlled definition of such a topic.\n\n    Each audio file has a corresponding text file, containing metadata such as author and tags. The dataset has been randomly split into 10 equal-size subsets. This is so that you can perform 10-fold crossvalidation in machine-learning experiments, or can use fixed subsets of the data (e.g. use one subset for development, and others for later validation). Each of the 10 subsets has about 128 minutes of audio; the dataset totals over 21 hours of audio.","metadata":{}},{"cell_type":"code","source":"def json_to_pd(file):\n    '''read and conv json file to pd row'''\n    with open(file) as f:\n        json_data = pd.json_normalize(json.loads(f.read()))\n        \n    return json_data\n\n\n\n\n#get all the json files (with description of the sounds)\nfile_list = Path(\"../input/freefield1010/freefield1010\").rglob(\"*.json\")\n\nall_audio = []\n\n\n\nfor filepath in file_list:\n    #conv json to pd \n    row = json_to_pd(filepath)\n    #add filepath to image()\n    row['filepath'] = str(filepath).rsplit('.',maxsplit=1)[0]  + '.wav'\n    \n    #append row to list\n    all_audio.append(row)\n    \n\n    \n    \nfreefield_df = pd.concat(all_audio,\n                         ignore_index=True)\n\nfreefield_df.head(3)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-06-09T14:30:18.993684Z","iopub.execute_input":"2024-06-09T14:30:18.994172Z","iopub.status.idle":"2024-06-09T14:30:59.649969Z","shell.execute_reply.started":"2024-06-09T14:30:18.994130Z","shell.execute_reply":"2024-06-09T14:30:59.648770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check if there is a bird call in audio\nfreefield_df['tags'] = freefield_df['tags'].apply(lambda x: x)\nfreefield_df['has_bird_call'] = freefield_df['tags'].apply(lambda x:(\"bird\" in x) or (\"birds\" in x) or(\"birdsong\" in x)).astype(int)\n\n#check number of birdcalls vs no calls in freefield data \nprint('Number of bird calls',freefield_df[freefield_df['has_bird_call']==1].shape[0])\nprint('Number of non bird calls',freefield_df[freefield_df['has_bird_call']!=1].shape[0])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-06-09T14:30:59.651561Z","iopub.execute_input":"2024-06-09T14:30:59.652045Z","iopub.status.idle":"2024-06-09T14:30:59.696276Z","shell.execute_reply.started":"2024-06-09T14:30:59.652002Z","shell.execute_reply":"2024-06-09T14:30:59.694851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Lets look at top tags**","metadata":{}},{"cell_type":"code","source":"\nall_tags = []\nfreefield_df['tags'].apply(lambda x: all_tags.extend(x))\npd.Series(all_tags).value_counts()[:20]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-06-09T14:30:59.700225Z","iopub.execute_input":"2024-06-09T14:30:59.700687Z","iopub.status.idle":"2024-06-09T14:30:59.754040Z","shell.execute_reply.started":"2024-06-09T14:30:59.700652Z","shell.execute_reply":"2024-06-09T14:30:59.752956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"non_bird_tags = [\"rain\",\"spring\",'forest','thunder','city','nature','wind','river','morning','night','sea',\"voice\"]\nnon_bird_freefield = freefield_df[((freefield_df['has_bird_call']!=1)&\n                                   (freefield_df['tags'].apply(lambda x: any([i in x for i in non_bird_tags]))))].reset_index(drop=True)\n\n#top selected tags\nall_tags = []\nnon_bird_freefield['tags'].apply(lambda x: all_tags.extend(x))\npd.Series(all_tags).value_counts()[:20]","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-06-09T14:30:59.755467Z","iopub.execute_input":"2024-06-09T14:30:59.755886Z","iopub.status.idle":"2024-06-09T14:30:59.827791Z","shell.execute_reply.started":"2024-06-09T14:30:59.755855Z","shell.execute_reply":"2024-06-09T14:30:59.826293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"freefield_unsam = freefield_df[freefield_df['has_bird_call']==1].reset_index()\nnon_bird_freefield = non_bird_freefield.sample(n= freefield_unsam.shape[0] * 2,random_state=CFG.RANDOM_SEED).copy()\n\n\nfreefield_unsam = pd.concat((freefield_unsam,non_bird_freefield),\n                                        ignore_index=True)\n\n#check number of birdcalls vs no calls in unsampled freefield data \nprint('Number of bird calls (unsampled)',freefield_unsam[freefield_unsam['has_bird_call']==1].shape[0])\nprint('Number of non bird calls (unsampled)',freefield_unsam[freefield_unsam['has_bird_call']!=1].shape[0])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-06-09T14:30:59.829311Z","iopub.execute_input":"2024-06-09T14:30:59.829742Z","iopub.status.idle":"2024-06-09T14:30:59.866315Z","shell.execute_reply.started":"2024-06-09T14:30:59.829709Z","shell.execute_reply":"2024-06-09T14:30:59.864939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"\n# Extracting Spectrograms\n\nFrom the blog here : https://www.macaulaylibrary.org/2021/07/19/from-sound-to-images-part-1-a-deep-dive-on-spectrogram-creation/","metadata":{}},{"cell_type":"markdown","source":"*What is a spectrogram*\n\n    A spectrogram tracks the sound frequencies (vertical axis) which appear in the waveform, as a function of time (horizontal axis). Brighter colors correspond to louder sounds.\n    \n*The reason to use spectrograms than direct sound*\n\n    Rather than working with waveforms directly, we have the option of representing our sound as an image either as a spectrogram or some other image representation. This approach has several advantages.\n    First, image representations are indispensable tools for bird sound ID experts who are trying to identify species in a recording.\n    Second, by using image representations, we have the option of using well-understood computer vision model architectures like RestNets along with pretrained weights from Imagenet.","metadata":{}},{"cell_type":"markdown","source":"# Parameters to consider when building spectrogram\n\n\nThere are a number of choices one can make when constructing a spectrogram. Among them are the following:\n\n• Clip length: How many seconds of audio should a spectrogram represent?\n\n> Shorter clips often eliminated important context from the soundscape,while longer clips are more expensive to train.\n\n\n• STFT window length: Represents a tradeoff between a spectrogram’s level of resolution in the time domain (short window length) and resolution in the frequency domain (long window length).\n\n> 256 or 512 samples \n\n\n• Mel scaling: In a frequency spectrogram, the vertical distance that represents an octave is not constant. As a result, it may be difficult for convolutional filters to learn to recognize harmonies, overtones, and repeated harmonic patterns. Mel scaling rescales the frequency axis, so that fixed differences in musical pitch (e.g. an octave or a fifth) correspond to fixed vertical distances. One possible downside to using mel scaling is that high frequency sounds will become compressed at the top of the spectrogram, and therefore might be harder to distinguish.\n\n> On or off\n\n\n• Image rescaling: Choices of the parameters above affect the spatial dimensions of the resulting spectrogram. To make meaningful comparisons, we chose a set of image dimensions to rescale our spectrograms to.\n\n>rescale to [128, 512] or [96, 512] (for hop size 128), or [128, 1024] (for hop size 64)","metadata":{}},{"cell_type":"markdown","source":"# Spectrogram Utils","metadata":{}},{"cell_type":"code","source":"#augmentations\naugmentations = Compose(\n    [\n            TimeStretch(min_rate=0.11,max_rate=0.3,p=0.25),\n            AddGaussianSNR(min_snr_in_db=5, max_snr_in_db=40, p=0.25)\n                        ]\n                        )\n\ndef plot_spec(path):\n    fig,ax = plt.subplots(figsize=(12,6))\n    \n    im = plt.imread(fname=path)\n    plt.axis('off')\n    plt.imshow(im,cmap='jet')\n    plt.colorbar(shrink=0.25)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:30:59.867863Z","iopub.execute_input":"2024-06-09T14:30:59.868221Z","iopub.status.idle":"2024-06-09T14:30:59.876164Z","shell.execute_reply.started":"2024-06-09T14:30:59.868190Z","shell.execute_reply":"2024-06-09T14:30:59.874847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#function to save stfts\n\ndef save_stft(file_path,\n              dir_path,\n              Augment=True,\n              ):\n    '''extracting mel-specs from given audio data and saving them to given folder'''\n    \n    #list to store spectogram and labels\n    stft_id = []\n        \n    #load audio     \n    sig,sr=librosa.load(file_path,\n                        sr=CFG.SAMPLE_RATE)\n    \n    \n    n=0\n    # break the signal into n second chunks\n    for i in range(0,len(sig),int(CFG.SIGNAL_LENGTH*CFG.SAMPLE_RATE)):\n        \n        window = sig[i:i + int(CFG.SIGNAL_LENGTH * CFG.SAMPLE_RATE)]\n\n        # End of signal\n        if len(window) < int(CFG.SIGNAL_LENGTH * CFG.SAMPLE_RATE):\n            break\n            \n            \n            \n        #Apply audio Augmentations :   \n        if Augment:\n            window=augmentations(window,\n                                 sample_rate=sr)\n            \n        # extracting mel-spectrograms:\n        mel_spec = librosa.feature.melspectrogram(y=window,\n                                                  sr=CFG.SAMPLE_RATE, \n                                                  n_fft=CFG.N_ffts,\n                                                  win_length=CFG.Window_size,\n                                                  hop_length=CFG.hop_length, \n                                                  n_mels=CFG.n_mels, \n                                                  fmin=CFG.FMIN, \n                                                  fmax=CFG.FMAX,\n                                                  norm=None\n                                                  )\n        \n        # log scaling (convert to decibels)\n        mel_spec = librosa.core.power_to_db(mel_spec,\n                                            ref=np.max)\n        # Normalize\n        mel_spec -= mel_spec.min()\n        mel_spec /= mel_spec.max()\n        \n        #saving Image\n        #image_id\n        ids=file_path.split('/')[-1].split('.')[0]\n        \n        save_id=f'{ids}_{n}.jpg'\n        save_path=os.path.join(dir_path,save_id)\n        n+=1\n        image = Image.fromarray(mel_spec * 255.0).convert(\"L\")\n        image.save(save_path)\n        \n        #saving_image ids and labels\n        stft_id.append(save_id)\n        \n    return stft_id","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:30:59.878193Z","iopub.execute_input":"2024-06-09T14:30:59.878947Z","iopub.status.idle":"2024-06-09T14:30:59.895343Z","shell.execute_reply.started":"2024-06-09T14:30:59.878906Z","shell.execute_reply":"2024-06-09T14:30:59.893934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Load data**","metadata":{}},{"cell_type":"code","source":"train_audio = '/kaggle/input/birdclef-2024/train_audio'\ntrain_metadata = pd.read_pickle('/kaggle/input/birdclef-2024-extenstive-eda/train_full.pkl')\n\n#make directories to save spectograms\nTrain_Spectrograms = './Train_Spectrograms'\nFreefield_Spectograms = './Freefield_Spectrograms'\n\n!mkdir $Train_Spectrograms\n!mkdir $Freefield_Spectograms","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:30:59.896962Z","iopub.execute_input":"2024-06-09T14:30:59.897403Z","iopub.status.idle":"2024-06-09T14:31:02.257428Z","shell.execute_reply.started":"2024-06-09T14:30:59.897371Z","shell.execute_reply":"2024-06-09T14:31:02.255775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extract training spectrograms","metadata":{}},{"cell_type":"code","source":"train_ids=[]\nfor idx,row in tqdm(train_metadata.iterrows()):\n    \n    #save spectograms\n    audio_ids = save_stft(file_path= row.filepath,\n                           dir_path =Train_Spectrograms)\n    \n    train_ids.extend(audio_ids)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:31:02.259958Z","iopub.execute_input":"2024-06-09T14:31:02.260412Z","iopub.status.idle":"2024-06-09T14:31:29.873155Z","shell.execute_reply.started":"2024-06-09T14:31:02.260374Z","shell.execute_reply":"2024-06-09T14:31:29.870511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**PLotting a few samples**","metadata":{}},{"cell_type":"code","source":"plot_spec(\"/kaggle/working/Train_Spectrograms/XC134896_0.jpg\")","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:32:35.576021Z","iopub.execute_input":"2024-06-09T14:32:35.576487Z","iopub.status.idle":"2024-06-09T14:32:35.930407Z","shell.execute_reply.started":"2024-06-09T14:32:35.576449Z","shell.execute_reply":"2024-06-09T14:32:35.929198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_spec(\"/kaggle/working/Train_Spectrograms/XC175797_1.jpg\")","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:32:59.765128Z","iopub.execute_input":"2024-06-09T14:32:59.765557Z","iopub.status.idle":"2024-06-09T14:33:00.174086Z","shell.execute_reply.started":"2024-06-09T14:32:59.765522Z","shell.execute_reply":"2024-06-09T14:33:00.172464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_spec(\"/kaggle/working/Train_Spectrograms/XC134896_0.jpg\")","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:31:43.663100Z","iopub.execute_input":"2024-06-09T14:31:43.663540Z","iopub.status.idle":"2024-06-09T14:31:44.950136Z","shell.execute_reply.started":"2024-06-09T14:31:43.663506Z","shell.execute_reply":"2024-06-09T14:31:44.948671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Joining metadata information with extracted spectrogram files**","metadata":{}},{"cell_type":"code","source":"train_df = pd.DataFrame(train_ids,\n                        columns=['spec_id'])\ntrain_df['file_id'] = train_df['spec_id'].apply(lambda x: x.split('_')[0])\n\n\ntrain_metadata['file_id'] = train_metadata['filename'].apply(lambda x: x.split('/')[1].split('.')[0])\n\n\n#join dfs\ntrain_df = train_df.merge(train_metadata,\n                          on='file_id',\n                          how='left')\n\n\ntrain_df.to_csv('train_df_24.csv',\n                index=False)\n\ntrain_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:31:29.881670Z","iopub.status.idle":"2024-06-09T14:31:29.882106Z","shell.execute_reply.started":"2024-06-09T14:31:29.881910Z","shell.execute_reply":"2024-06-09T14:31:29.881928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Extract Freefield spectrograms","metadata":{}},{"cell_type":"code","source":"freefield_ids=[]\nfor idx, row in tqdm(freefield_unsam.iterrows()):\n    \n    #save spectograms\n    audio_ids = save_stft(file_path= row.filepath,\n                           dir_path =Freefield_Spectograms)\n    \n    freefield_ids.extend(audio_ids)\n    \nprint('Number of samples extracted ',len(freefield_ids))","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:33:13.304115Z","iopub.execute_input":"2024-06-09T14:33:13.304518Z","iopub.status.idle":"2024-06-09T14:33:17.928232Z","shell.execute_reply.started":"2024-06-09T14:33:13.304489Z","shell.execute_reply":"2024-06-09T14:33:17.926096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"freefield_ids = pd.DataFrame(freefield_ids,\n                            columns=['spec_id'])\nfreefield_ids['file_id'] = freefield_ids['spec_id'].apply(lambda x: x.split('_')[0])\nfreefield_ids","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:33:27.894881Z","iopub.execute_input":"2024-06-09T14:33:27.895296Z","iopub.status.idle":"2024-06-09T14:33:27.912214Z","shell.execute_reply.started":"2024-06-09T14:33:27.895266Z","shell.execute_reply":"2024-06-09T14:33:27.910779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_spec(\"/kaggle/working/Freefield_Spectrograms/40822_0.jpg\")","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:33:31.964188Z","iopub.execute_input":"2024-06-09T14:33:31.964577Z","iopub.status.idle":"2024-06-09T14:33:32.373931Z","shell.execute_reply.started":"2024-06-09T14:33:31.964547Z","shell.execute_reply":"2024-06-09T14:33:32.372661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_spec(\"/kaggle/working/Freefield_Spectrograms/122766_0.jpg\")","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:33:34.400390Z","iopub.execute_input":"2024-06-09T14:33:34.400874Z","iopub.status.idle":"2024-06-09T14:33:34.821424Z","shell.execute_reply.started":"2024-06-09T14:33:34.400836Z","shell.execute_reply":"2024-06-09T14:33:34.820273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_spec(\"/kaggle/working/Freefield_Spectrograms/157404_0.jpg\")","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:33:36.024664Z","iopub.execute_input":"2024-06-09T14:33:36.025213Z","iopub.status.idle":"2024-06-09T14:33:36.431242Z","shell.execute_reply.started":"2024-06-09T14:33:36.025158Z","shell.execute_reply":"2024-06-09T14:33:36.429859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#saving free field df \nfreefield_unsam.to_csv('freefield_downsampled.csv',index=False)\nfreefield_ids.to_csv(\"freefield_downsampled_specs.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T14:31:29.894138Z","iopub.status.idle":"2024-06-09T14:31:29.894531Z","shell.execute_reply.started":"2024-06-09T14:31:29.894346Z","shell.execute_reply":"2024-06-09T14:31:29.894363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Thank you. The next and previous notebooks in this series are in mentioned in the top markdown section of this notebook**","metadata":{}}]}