{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Preprocessing Data \n","metadata":{}},{"cell_type":"markdown","source":"## 1. Import the libraries","metadata":{}},{"cell_type":"code","source":"from IPython.display import Audio \nimport numpy as np \nimport matplotlib.pyplot as plt \nimport PIL\nimport librosa\nimport os \nimport torch\nfrom torch.utils.data import random_split \nfrom torch.utils.data import Dataset\nimport torchaudio\nimport pandas as pd \nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:45:16.618823Z","iopub.execute_input":"2023-06-20T11:45:16.619424Z","iopub.status.idle":"2023-06-20T11:45:21.206739Z","shell.execute_reply.started":"2023-06-20T11:45:16.619367Z","shell.execute_reply":"2023-06-20T11:45:21.205293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Set up parameters for the project\n\nSet up Parameters for processing audio and working with MFCC transforms","metadata":{}},{"cell_type":"code","source":"SAMPLE_RATE = 22050\nN_FFT = 1024 \nHOP_LENGTH = 512 \nN_MFCC = 40 \nCLIP_LENGTH = 15\nNUM_SAMPLES = SAMPLE_RATE * CLIP_LENGTH\n\n# Set device for pytorch\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:45:21.209816Z","iopub.execute_input":"2023-06-20T11:45:21.210580Z","iopub.status.idle":"2023-06-20T11:45:21.218456Z","shell.execute_reply.started":"2023-06-20T11:45:21.210537Z","shell.execute_reply":"2023-06-20T11:45:21.216524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Set up directories for facilitating import of data and additional files","metadata":{}},{"cell_type":"code","source":"ROOT_DIR = '/kaggle/input/birdclef-2023/'\nAUDIO_DIR = ROOT_DIR+'train_audio/'\nMETADATA_DIR = ROOT_DIR+'train_metadata.csv'\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:45:21.220627Z","iopub.execute_input":"2023-06-20T11:45:21.221492Z","iopub.status.idle":"2023-06-20T11:45:21.229129Z","shell.execute_reply.started":"2023-06-20T11:45:21.221419Z","shell.execute_reply":"2023-06-20T11:45:21.227820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Check out the taxonomy file\n\nCheck out the taxonomy file and find out if there are any useful items in the csv file","metadata":{}},{"cell_type":"code","source":"TAXONOMY = pd.read_csv('/kaggle/input/birdclef-2023/eBird_Taxonomy_v2021.csv')\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:45:21.231275Z","iopub.execute_input":"2023-06-20T11:45:21.231757Z","iopub.status.idle":"2023-06-20T11:45:21.368344Z","shell.execute_reply.started":"2023-06-20T11:45:21.231718Z","shell.execute_reply":"2023-06-20T11:45:21.366991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Load the Metadata file\n\nLoad the metadata file to see the labels and individual files <br>\nColumn for -<br>\nLabels: primary_label <br>\nFiles: filename<br>","metadata":{}},{"cell_type":"code","source":"METADATA = pd.read_csv(ROOT_DIR+'train_metadata.csv')\nMETADATA.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:45:21.372047Z","iopub.execute_input":"2023-06-20T11:45:21.372487Z","iopub.status.idle":"2023-06-20T11:45:21.589330Z","shell.execute_reply.started":"2023-06-20T11:45:21.372450Z","shell.execute_reply":"2023-06-20T11:45:21.588016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(METADATA.shape)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:45:21.591431Z","iopub.execute_input":"2023-06-20T11:45:21.591954Z","iopub.status.idle":"2023-06-20T11:45:21.600135Z","shell.execute_reply.started":"2023-06-20T11:45:21.591890Z","shell.execute_reply":"2023-06-20T11:45:21.598868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are a total of **16941** files in the dataset","metadata":{}},{"cell_type":"code","source":"label_list = os.listdir(AUDIO_DIR)\nprint(len(label_list))","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:45:21.601789Z","iopub.execute_input":"2023-06-20T11:45:21.602287Z","iopub.status.idle":"2023-06-20T11:45:21.664529Z","shell.execute_reply.started":"2023-06-20T11:45:21.602241Z","shell.execute_reply":"2023-06-20T11:45:21.663001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are a total of **264** classes of audio data","metadata":{}},{"cell_type":"markdown","source":"## 5. Get the MFCC representation of Audio file using <span style=\"color: red;\">Librosa</span>\n\nDepending on the preference we can use either librosa or torchaudio for loading audio data. The *AudioProcessor* class is used to load files and process transformations based on librosa.","metadata":{}},{"cell_type":"code","source":"class AudioProcessor:\n    def __init__(self):\n        self.info = \"Audio Processor Class for various audio processor operations\"\n    \n    \n    def open_audio_file(self, filepath):\n        audio, sample_rate = librosa.load(filepath)\n        return audio, sample_rate\n        \n        \n    def audio_resample(self, y, orig_sr, target_sr):\n        rs = librosa.resample(y=y, orig_sr=orig_sr, target_sr = target_sr)\n        return rs \n        \n        \n    def get_mfcc(self,audio,sr = 32000, n_mfcc=32, **kwargs ):\n            \n        mfcc_data = librosa.feature.mfcc(y = audio, sr = sr,  n_mfcc = n_mfcc, **kwargs)\n        \n        return mfcc_data\n    \n    def preview_audio(self, audio):\n        return Audio(audio)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:45:21.666448Z","iopub.execute_input":"2023-06-20T11:45:21.666874Z","iopub.status.idle":"2023-06-20T11:45:21.680498Z","shell.execute_reply.started":"2023-06-20T11:45:21.666840Z","shell.execute_reply":"2023-06-20T11:45:21.679066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. Class to add additional helper functions for data processing\nThe *Data Helper* class adds additional functions for getting data ","metadata":{}},{"cell_type":"code","source":"\n\nclass DataHelper:\n    \n    def data_list(self,datadir,metadata):\n        audio_list = [datadir + x  for x in list(metadata['filename'])]\n        labels = [y for y in list(metadata['primary_label'])]\n        return audio_list, labels\n    \n    # For parsing with librosa/Have used torchaudio instead/Kept this for funzies\n#     def parse_audio(self,audio_list):\n#         audio_processor = AudioProcessor()\n        \n#         for audio_file in audio_list:\n#             single_audio_file, sr = audio_processor.open_audio_file(audio_file)\n#             resampled_file = audio_processor.audio_resample(single_audio_file, sr, 32000)\n            \n#             pass\n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:45:21.682103Z","iopub.execute_input":"2023-06-20T11:45:21.682478Z","iopub.status.idle":"2023-06-20T11:45:21.699325Z","shell.execute_reply.started":"2023-06-20T11:45:21.682438Z","shell.execute_reply":"2023-06-20T11:45:21.698077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 7. Create Custom Dataset Loader for <span style=\"color: red;\">Pytorch</span>\n\nUsing <span style=\"color: red;\">*torchaudio*</span> and Pytorch to create a custom dataloader.","metadata":{}},{"cell_type":"code","source":"\nclass AudioDataset(Dataset):\n    def __init__(self, annotations, \n                 root_dir,\n                 num_samples = 16000,\n                 transform=None, \n                 target_sample_rate = 16000):\n        self.root_dir = root_dir\n        self.annotations = pd.read_csv(annotations)\n        self.transfrom = transform\n        self.target_sample_rate = target_sample_rate\n        self.num_samples = num_samples\n        \n    def __len__(self):\n        \"\"\"\n        Returns the length of the Dataset/filelist input\n        \"\"\"\n        return len(self.annotations)\n    \n    def __getitem__(self, idx):\n        # readfile\n        audio_sample_path = self._get_audio_sample_path(index)\n        label = self._get_audio_sample_label(index)\n        signal, sr = torchaudio.load(audio_sample_path)\n#         signal = signal.to(device)\n        # apply transform\n        signal = self._resample_if_necessary(signal,sr)\n        signal = self._mix_down_if_necessary(signal) #if stereo audio convert to mono\n        signal = self._cut_audio_if_necessary(signal)\n        signal = self._right_pad_if_necessary(signal)\n        signal = self.transform(signal)\n        return signal, label\n    \n    def _right_pad_if_necessary(self, signal):\n        if signal.shape[1] < self.num_samples:\n            missing_samples = self.num_samples - signal.shape[1]\n            last_dim_padding = (0, missing_samples)\n            signal = torch.nn.functional.pad(signal, last_dim_padding)\n        return signal \n    \n    def _mix_down_if_necessary(self, signal):\n        if signal.shape[1] > self.num_samples:\n            signal = signal[:,:self.num_samples]\n        return signal\n    \n    \n    def _get_audio_sample_path(self, index):\n        path = self.root_dir + self.annotations.iloc[index].filename\n        return path\n    \n    def _get_audio_sample_label(self, index):\n        return self.annotations.iloc[index].primary_label     \n    \n    def _resample_if_necessary(self,signal, sr):\n        if sr!=self.target_sample_rate:\n            resampler = torchaudio.transforms.Resample(sr, self.target_sample_rate)\n            signal = resampler(signal)\n        return signal\n    \n    def _mix_down_if_necessary(self, signal):\n        if signal.shape[0] >1:\n            signal = torch.mean(signal, dim=0,keepdim=True)\n        return signal\n        \n        ","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:45:21.700881Z","iopub.execute_input":"2023-06-20T11:45:21.701450Z","iopub.status.idle":"2023-06-20T11:45:21.722994Z","shell.execute_reply.started":"2023-06-20T11:45:21.701401Z","shell.execute_reply":"2023-06-20T11:45:21.721634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 8. Set up an audio processing pipeline using torchaudio","metadata":{}},{"cell_type":"code","source":"# Get all the audio files list\ndataParse = DataHelper()\naudio_list, labels = dataParse.data_list(AUDIO_DIR, METADATA)\n\n\n# Get the audio file with maximum duration\n\n\n\n# Set up mfcc transformation to be passed\nmfcc_transform = torchaudio.transforms.MFCC(sample_rate = SAMPLE_RATE,\n                                           n_mfcc = N_MFCC,\n                                           dct_type = 2,\n                                           norm='ortho',\n                                           log_mels = False)\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:45:21.724829Z","iopub.execute_input":"2023-06-20T11:45:21.725344Z","iopub.status.idle":"2023-06-20T11:45:21.884036Z","shell.execute_reply.started":"2023-06-20T11:45:21.725292Z","shell.execute_reply":"2023-06-20T11:45:21.882835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audioDatasetLoader = AudioDataset(METADATA_DIR, \n                                  AUDIO_DIR,num_samples=NUM_SAMPLES,\n                                  transform= mfcc_transform, \n                                  target_sample_rate=SAMPLE_RATE,\n                                  )\n\n","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:45:21.885630Z","iopub.execute_input":"2023-06-20T11:45:21.886665Z","iopub.status.idle":"2023-06-20T11:45:21.995632Z","shell.execute_reply.started":"2023-06-20T11:45:21.886617Z","shell.execute_reply":"2023-06-20T11:45:21.993894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 9. Fixing Data Variability\n* > Check the data and find out clip durations \n* > Find out the variability in audio length\n* > Pad the audio that is of shorter length\n* > Trim long audio\n","metadata":{}},{"cell_type":"code","source":"\n# Check the length of audio clips\nclip_length = []\nfor item in tqdm(audio_list):\n    test_sig, test_sr = librosa.load(item)\n    time = librosa.get_duration(y = test_sig,sr = test_sr )\n    del test_sig\n    clip_length.append(time)\n\nprint(clip_length[0:50])","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:45:21.997852Z","iopub.execute_input":"2023-06-20T11:45:21.998556Z","iopub.status.idle":"2023-06-20T12:08:56.871822Z","shell.execute_reply.started":"2023-06-20T11:45:21.998506Z","shell.execute_reply":"2023-06-20T12:08:56.869960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Save the clip lengths as numpy array so it is not needed to reprocess in the future\n\n# with open(ROOT_DIR+'clip_length.npy', 'w') as f:\n#     np.save(f, np.array(clip_length))","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:08:56.878204Z","iopub.execute_input":"2023-06-20T12:08:56.880620Z","iopub.status.idle":"2023-06-20T12:08:56.886864Z","shell.execute_reply.started":"2023-06-20T12:08:56.880540Z","shell.execute_reply":"2023-06-20T12:08:56.885722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Find the length of clips and find the mean value for the clip duration","metadata":{}},{"cell_type":"code","source":"print(\"Minimum clip length: \", min(clip_length))\nprint(\"Maximum clip length:  \", max(clip_length))\nprint(\"Average Clip Length: \",sum(clip_length)/len(clip_length))\n\n# Clips less than 1 second\nl1 = sum( item<= 1.0 for item in clip_length)\nprint(\"Clips less than 1 second: \",l1)\n\n# Clips less than 5 seconds\nl5 = sum (item<=5.0 for item in clip_length)\nprint(\"less than 5 second: \", l5)\n# Clips less than 10 second\nl10 = sum (item<=10.0 for item in clip_length)\nprint(\"Less than 10 seconds: \",l10)\n# Clips more than 20 second\nm20 = len([i for i in clip_length if i>=20.0])\nprint(\"More than 20 seconds: \", m20)\n\nm30 = len([i for i in clip_length if i>=30.0])\nprint(\"More than 30 seconds: \", m30)\n\nm40 = len([i for i in clip_length if i>40.0])","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:08:56.888476Z","iopub.execute_input":"2023-06-20T12:08:56.888891Z","iopub.status.idle":"2023-06-20T12:08:56.919287Z","shell.execute_reply.started":"2023-06-20T12:08:56.888858Z","shell.execute_reply":"2023-06-20T12:08:56.917222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(clip_length, bins=45)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:08:56.921195Z","iopub.execute_input":"2023-06-20T12:08:56.921581Z","iopub.status.idle":"2023-06-20T12:08:57.308235Z","shell.execute_reply.started":"2023-06-20T12:08:56.921550Z","shell.execute_reply":"2023-06-20T12:08:57.306854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 10. Find out the number of <span style=\"color: red;\">data points in each class</span> and plot it\n","metadata":{}},{"cell_type":"code","source":"class_counts = METADATA.primary_label.value_counts()\nplt.plot(class_counts)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:08:57.309815Z","iopub.execute_input":"2023-06-20T12:08:57.310193Z","iopub.status.idle":"2023-06-20T12:08:59.360234Z","shell.execute_reply.started":"2023-06-20T12:08:57.310162Z","shell.execute_reply":"2023-06-20T12:08:59.359029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"There are still a number of interesting things that we can do with the dataset but I just wanted to publish it for people who want to get started with the audio processing part. A number of helper functions are present in the **AudioProcessor** class for further processing or to convert the data into their MFCC representation","metadata":{}}]}