{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<h1 align='center'> BirdCLEF Starter Notebook </h1>","metadata":{}},{"cell_type":"markdown","source":"## STEP 1 : Installations and Imports","metadata":{}},{"cell_type":"code","source":"import os \nimport pandas as pd \npd.set_option('display.max_columns',300)\nimport warnings\nwarnings.filterwarnings('ignore')\nimport numpy as np \nimport matplotlib.pyplot as plt \nimport seaborn as sns \nimport IPython.display as ipd\nimport librosa\nimport librosa.display\nimport wave\nfrom scipy.io import wavfile\nimport tensorflow as tf\nimport tensorflow_io as tfio\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:48.898279Z","iopub.execute_input":"2023-03-11T10:01:48.898747Z","iopub.status.idle":"2023-03-11T10:01:48.906839Z","shell.execute_reply.started":"2023-03-11T10:01:48.898707Z","shell.execute_reply":"2023-03-11T10:01:48.905653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## STEP 2 : Auralize & Visualize Data","metadata":{}},{"cell_type":"code","source":"main_dir = '/kaggle/input/birdclef-2023/train_audio'\nclass_labels= sorted(os.listdir(main_dir))\nprint(\"Total number of bird audio class : \",len(class_labels))","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:49.002626Z","iopub.execute_input":"2023-03-11T10:01:49.003721Z","iopub.status.idle":"2023-03-11T10:01:49.012853Z","shell.execute_reply.started":"2023-03-11T10:01:49.003672Z","shell.execute_reply":"2023-03-11T10:01:49.011567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Among the 264 audio sample classes we will try to auralize and visualize the waveforms of 2 classes of bird species.","metadata":{}},{"cell_type":"code","source":"class1_path = os.path.join(main_dir, class_labels[0])\naudio_of_class_1 = os.listdir(class1_path)[0]\nclass1_audio_path = os.path.join(class1_path, audio_of_class_1)\nprint(\"Class 1 Label : \", class1_path.split('/')[-1])\n\n\nclass2_path = os.path.join(main_dir, class_labels[-1])\naudio_of_class_2 = os.listdir(class2_path)[0]\nclass2_audio_path = os.path.join(class2_path, audio_of_class_2)\nprint(\"Class 2 Label : \", class2_path.split('/')[-1])","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:49.014899Z","iopub.execute_input":"2023-03-11T10:01:49.015399Z","iopub.status.idle":"2023-03-11T10:01:49.026706Z","shell.execute_reply.started":"2023-03-11T10:01:49.015347Z","shell.execute_reply":"2023-03-11T10:01:49.025214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***1. abethr1 audio sample***","metadata":{}},{"cell_type":"code","source":"ipd.Audio(class1_audio_path, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:49.117837Z","iopub.execute_input":"2023-03-11T10:01:49.118591Z","iopub.status.idle":"2023-03-11T10:01:49.139123Z","shell.execute_reply.started":"2023-03-11T10:01:49.118532Z","shell.execute_reply":"2023-03-11T10:01:49.137619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data, sr = librosa.load(class1_audio_path)\nplt.figure(figsize=(20,6))\nplt.title(\"Waveform of abethr1 audio sample\")\nlibrosa.display.waveshow(data, sr=sr,color='#A6EAFF')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:49.141648Z","iopub.execute_input":"2023-03-11T10:01:49.142087Z","iopub.status.idle":"2023-03-11T10:01:49.748613Z","shell.execute_reply.started":"2023-03-11T10:01:49.142046Z","shell.execute_reply":"2023-03-11T10:01:49.747582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***2.  yewgre1 audio sample***","metadata":{}},{"cell_type":"code","source":"ipd.Audio(class2_audio_path, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:49.749819Z","iopub.execute_input":"2023-03-11T10:01:49.750864Z","iopub.status.idle":"2023-03-11T10:01:49.764438Z","shell.execute_reply.started":"2023-03-11T10:01:49.750825Z","shell.execute_reply":"2023-03-11T10:01:49.762978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data, sr = librosa.load(class2_audio_path)\nplt.figure(figsize=(20,6))\nplt.title(\"Waveform of yewgre1 audio sample\")\nlibrosa.display.waveshow(data, sr=sr,color='#A6EAFF')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:49.767666Z","iopub.execute_input":"2023-03-11T10:01:49.768490Z","iopub.status.idle":"2023-03-11T10:01:50.657657Z","shell.execute_reply.started":"2023-03-11T10:01:49.768421Z","shell.execute_reply":"2023-03-11T10:01:50.656415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## STEP 3 : Understanding the audio data","metadata":{}},{"cell_type":"code","source":"### Hearing out the test data\nipd.Audio('/kaggle/input/birdclef-2023/test_soundscapes/soundscape_29201.ogg', rate=16000)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:50.658872Z","iopub.execute_input":"2023-03-11T10:01:50.659203Z","iopub.status.idle":"2023-03-11T10:01:50.794640Z","shell.execute_reply.started":"2023-03-11T10:01:50.659148Z","shell.execute_reply":"2023-03-11T10:01:50.793750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bird_df=pd.read_csv(\"/kaggle/input/birdclef-2023/eBird_Taxonomy_v2021.csv\")\ntrain_meta_df = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\nsample_submission_df = pd.read_csv('/kaggle/input/birdclef-2023/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:50.795744Z","iopub.execute_input":"2023-03-11T10:01:50.796277Z","iopub.status.idle":"2023-03-11T10:01:50.940792Z","shell.execute_reply.started":"2023-03-11T10:01:50.796242Z","shell.execute_reply":"2023-03-11T10:01:50.939317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:50.942203Z","iopub.execute_input":"2023-03-11T10:01:50.942558Z","iopub.status.idle":"2023-03-11T10:01:51.039877Z","shell.execute_reply.started":"2023-03-11T10:01:50.942524Z","shell.execute_reply":"2023-03-11T10:01:51.038709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So from the sample submission dataset it becomes clear thet we would have to:\n1. Slice the 10 min long \"test_soundscapes\" data into 5 second window.\n2. On that sliced 5 sec window we need to find the probabilities of each classes.\n3. It's a Multiclass classification problem with number of classes = 264","metadata":{}},{"cell_type":"code","source":"def getAudioInfo(main_dir, class_labels):\n    '''\n        The function will iterate of all the audio files in all the classes to extract \n        information from the audio files which will help us to decide what pre-processing\n        steps are to be implemented in the future.\n    '''\n    df_list=[]\n    for i in range(len(class_labels)):\n        df=pd.DataFrame()\n        class_path = os.path.join(main_dir, class_labels[i])\n        audio_files = os.listdir(class_path)\n        for x in range(len(audio_files)):\n            filename= audio_files[x]\n            filepath= os.path.join(class_path, filename)\n            sr = tfio.audio.AudioIOTensor(filepath).rate.numpy()\n            data= tfio.audio.AudioIOTensor(filepath).to_tensor()\n            waveform_length  = data.shape[0]\n            n_channel = data.shape[1]\n            audio_duration = waveform_length/sr\n            df.loc[x,'primary_label']=class_labels[i]\n            df.loc[x,'filename'] = filename\n            df.loc[x,'wavform_length'] = waveform_length\n            df.loc[x,'sample_rate'] = sr\n            df.loc[x,'n_channel'] = n_channel\n            df.loc[x,'audio_duration'] = audio_duration\n            \n        df_list.append(df)\n        del df\n    return df_list","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:51.041226Z","iopub.execute_input":"2023-03-11T10:01:51.041576Z","iopub.status.idle":"2023-03-11T10:01:51.054800Z","shell.execute_reply.started":"2023-03-11T10:01:51.041542Z","shell.execute_reply":"2023-03-11T10:01:51.053534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Commenting the following cells as running them will require 20 mins to save the notebook for publishing.","metadata":{}},{"cell_type":"code","source":"# %%time\n# df_list= getAudioInfo(main_dir, class_labels)\n# audio_df=pd.concat(df_list_1+df_list_2, axis=0).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:51.057814Z","iopub.execute_input":"2023-03-11T10:01:51.058882Z","iopub.status.idle":"2023-03-11T10:01:51.067537Z","shell.execute_reply.started":"2023-03-11T10:01:51.058828Z","shell.execute_reply":"2023-03-11T10:01:51.066140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Saving the audio dataframe that we extracted from the audio info\n# audio_df.to_csv('audio_info_df.csv')","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:51.071376Z","iopub.execute_input":"2023-03-11T10:01:51.072058Z","iopub.status.idle":"2023-03-11T10:01:51.080448Z","shell.execute_reply.started":"2023-03-11T10:01:51.072016Z","shell.execute_reply":"2023-03-11T10:01:51.079058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_df = pd.read_csv('/kaggle/input/birdclef-audio-info/audio_info_df.csv').drop('Unnamed: 0', axis=1)\naudio_df","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:51.081758Z","iopub.execute_input":"2023-03-11T10:01:51.082097Z","iopub.status.idle":"2023-03-11T10:01:51.133754Z","shell.execute_reply.started":"2023-03-11T10:01:51.082064Z","shell.execute_reply":"2023-03-11T10:01:51.132527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## STEP 4 : EDA","metadata":{}},{"cell_type":"code","source":"grouped_audio_df_1 = audio_df.groupby(by='primary_label')[['primary_label']].count().rename(columns={'primary_label':'label_counts'}).sort_values('label_counts', ascending=False).reset_index()\ngrouped_audio_df_2 = audio_df.groupby(by='primary_label')[['audio_duration']].mean().rename(columns={'audio_duration':'mean_audio_duration'}).sort_values('mean_audio_duration', ascending=False).reset_index()\ngrouped_audio_df = pd.merge(left=grouped_audio_df_1, right=grouped_audio_df_2, how='left', on='primary_label')","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:51.135403Z","iopub.execute_input":"2023-03-11T10:01:51.136227Z","iopub.status.idle":"2023-03-11T10:01:51.159925Z","shell.execute_reply.started":"2023-03-11T10:01:51.136189Z","shell.execute_reply":"2023-03-11T10:01:51.158888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grouped_audio_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:51.161084Z","iopub.execute_input":"2023-03-11T10:01:51.161418Z","iopub.status.idle":"2023-03-11T10:01:51.173794Z","shell.execute_reply.started":"2023-03-11T10:01:51.161385Z","shell.execute_reply":"2023-03-11T10:01:51.172400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grouped_audio_df .tail(5)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:51.175967Z","iopub.execute_input":"2023-03-11T10:01:51.176409Z","iopub.status.idle":"2023-03-11T10:01:51.188890Z","shell.execute_reply.started":"2023-03-11T10:01:51.176354Z","shell.execute_reply":"2023-03-11T10:01:51.187554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above dataframe depicts that there are only 1 sample audio in some of the classes and in some labels there are 500 samples so the non-uniformity of the samples withing the classes would make it difficult for the models to predict. So the class imbalance needs to be handled. To be sure of the info being correct, lets cross check once:","metadata":{}},{"cell_type":"code","source":"## Cross-check : Applying of filter returns only one sample as expected\naudio_df.loc[audio_df['primary_label']=='crefra2']","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:51.190603Z","iopub.execute_input":"2023-03-11T10:01:51.190947Z","iopub.status.idle":"2023-03-11T10:01:51.206812Z","shell.execute_reply.started":"2023-03-11T10:01:51.190911Z","shell.execute_reply":"2023-03-11T10:01:51.205500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Cross-check : Here we see the duration is 41 sec\nclassname = audio_df.loc[audio_df['primary_label']=='crefra2'].values[0][0]\nfilename = audio_df.loc[audio_df['primary_label']=='crefra2'].values[0][1]\nipd.Audio(main_dir+'/'+classname+'/'+filename, rate=16000)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:51.208104Z","iopub.execute_input":"2023-03-11T10:01:51.208454Z","iopub.status.idle":"2023-03-11T10:01:51.230789Z","shell.execute_reply.started":"2023-03-11T10:01:51.208421Z","shell.execute_reply":"2023-03-11T10:01:51.229589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of channels in all the audio clips : \", audio_df.n_channel.nunique())\nprint(\"Number of unique sample rates in all the clips : \", audio_df.sample_rate.nunique())","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:51.232267Z","iopub.execute_input":"2023-03-11T10:01:51.232657Z","iopub.status.idle":"2023-03-11T10:01:51.239478Z","shell.execute_reply.started":"2023-03-11T10:01:51.232605Z","shell.execute_reply":"2023-03-11T10:01:51.238199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Distribution of the average audio duration across classes**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16,5))\nsns.distplot(grouped_audio_df['mean_audio_duration'])\nplt.title(\"Distribution of average audio duration across classes\")\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:51.241041Z","iopub.execute_input":"2023-03-11T10:01:51.242154Z","iopub.status.idle":"2023-03-11T10:01:51.710451Z","shell.execute_reply.started":"2023-03-11T10:01:51.242102Z","shell.execute_reply":"2023-03-11T10:01:51.709483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Distribution of audio sample counts across classes**","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16,5))\nsns.countplot(x= audio_df['primary_label'])\nplt.tick_params(axis='x',bottom=False, top=False,labelbottom=False)\nplt.title(\"Count of audio samples across classes\")\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2023-03-11T10:01:51.711723Z","iopub.execute_input":"2023-03-11T10:01:51.712571Z","iopub.status.idle":"2023-03-11T10:01:54.076660Z","shell.execute_reply.started":"2023-03-11T10:01:51.712530Z","shell.execute_reply":"2023-03-11T10:01:54.075408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<h5 align='right'>........................... To be continued</h5>","metadata":{}},{"cell_type":"markdown","source":"<h5 align='center'> Please Upvote, if you find the notebook useful : )</h5>","metadata":{}},{"cell_type":"markdown","source":"<h3 align='center'> Thank you ; )</h3>","metadata":{}}]}