{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#importing required packages\nimport pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\nimport os\nimport multiprocessing\n\nfrom pydub import AudioSegment\nfrom IPython.display import Audio\nimport librosa\nimport librosa.display\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-21T09:45:50.066875Z","iopub.execute_input":"2023-03-21T09:45:50.068039Z","iopub.status.idle":"2023-03-21T09:45:50.075198Z","shell.execute_reply.started":"2023-03-21T09:45:50.067990Z","shell.execute_reply":"2023-03-21T09:45:50.073322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#defining base dir paths\naudio_path = '/kaggle/input/birdclef-2023/train_audio'\nmeta_path = '/kaggle/input/birdclef-2023/train_metadata.csv'\ntaxonomy_path = '/kaggle/input/birdclef-2023/eBird_Taxonomy_v2021.csv'","metadata":{"execution":{"iopub.status.busy":"2023-03-21T07:42:14.160477Z","iopub.execute_input":"2023-03-21T07:42:14.161388Z","iopub.status.idle":"2023-03-21T07:42:14.166413Z","shell.execute_reply.started":"2023-03-21T07:42:14.161348Z","shell.execute_reply":"2023-03-21T07:42:14.165171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### **Exploratory Data Analysis**","metadata":{}},{"cell_type":"code","source":"df = pd.DataFrame()\n\nmeta_df = pd.read_csv(meta_path)\ndf['class'] = meta_df['primary_label']\ndf['path'] = audio_path + '/' + meta_df['filename'] #absolute path to the audio file in the dataset\n\ndf.sample(5)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T07:42:14.184725Z","iopub.execute_input":"2023-03-21T07:42:14.185782Z","iopub.status.idle":"2023-03-21T07:42:14.278455Z","shell.execute_reply.started":"2023-03-21T07:42:14.185712Z","shell.execute_reply":"2023-03-21T07:42:14.276792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Number of unique spcies: ', len(df['class'].unique()))\nprint(\"Unique species: \\n\", df['class'].unique() )","metadata":{"execution":{"iopub.status.busy":"2023-03-21T07:42:14.280551Z","iopub.execute_input":"2023-03-21T07:42:14.280964Z","iopub.status.idle":"2023-03-21T07:42:14.291595Z","shell.execute_reply.started":"2023-03-21T07:42:14.280923Z","shell.execute_reply":"2023-03-21T07:42:14.290112Z"},"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class distribution\nplt.figure(figsize=(10, 50)) # width, height (inches)\n\nsns.countplot(y = df['class'])\nplt.title('Class distribution of the dataset')\nplt.xlabel('Number of bird calls')\nplt.ylabel('Species name')\nplt.show()\n\n#Inference:\n#there is a class imbalance in the dataset, few class have 500 audio recordings and some have audio samples in single digits\n#we have consider this while splitting and training the model","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2023-03-21T07:58:51.999527Z","iopub.execute_input":"2023-03-21T07:58:52.000441Z","iopub.status.idle":"2023-03-21T07:58:55.734291Z","shell.execute_reply.started":"2023-03-21T07:58:52.000395Z","shell.execute_reply":"2023-03-21T07:58:55.732776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# binning the dataset based on number of audio samples avaiable\nplt.figure(figsize=(10, 5)) # width, height (inches)\n\ntemp = df['class'].value_counts()\nax = sns.histplot(temp, binwidth=100)\ntotals = []\nfor p in ax.patches:\n    totals.append(p.get_height())\ntotal = sum(totals)\nfor p in ax.patches:\n    percentage = '{:.1f}%'.format(100 * p.get_height()/total)\n    x = p.get_x() + (p.get_width() / 2)\n    y = p.get_y() + p.get_height() + 0.01\n    ax.annotate(percentage, (x, y))\n\nplt.xlabel('No of audio samples available for a bird spcies')\nplt.show()\n\n\n#Inference:\n#We observe that 83% of bird species have audio samples of range 0 to 100 - which is very low for a deep learning models","metadata":{"execution":{"iopub.status.busy":"2023-03-21T09:01:08.032063Z","iopub.execute_input":"2023-03-21T09:01:08.032467Z","iopub.status.idle":"2023-03-21T09:01:08.289975Z","shell.execute_reply.started":"2023-03-21T09:01:08.032430Z","shell.execute_reply":"2023-03-21T09:01:08.288645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Going farther deeper to see how many bird species have very number of audio samples to train on\nplt.figure(figsize=(10, 5)) # width, height (inches)\n\ntemp = df['class'].value_counts()\nax = sns.histplot(temp, binwidth=20)\ntotals = []\nfor p in ax.patches:\n    totals.append(p.get_height())\ntotal = sum(totals)\nfor p in ax.patches:\n    percentage = '{:.1f}%'.format(100 * p.get_height()/total)\n    x = p.get_x() + (p.get_width() / 2)\n    y = p.get_y() + p.get_height() + 0.01\n    ax.annotate(percentage, (x, y))\n\nplt.xlabel('No of audio samples available for a bird spcies')\nplt.show()\n\n#Inference: \n# when bindwidth = 20; around 36% bird species that is approx (0.36*264) = 95 birds have less than 20 audio samples to train on\n# and 25% ar in range in 20 to 40 samples and so on.\n# when bindwidth = 1; around 2% bird species that is approx (0.02*264) = 5 birds have only 1 audio sample in dataset.","metadata":{"execution":{"iopub.status.busy":"2023-03-21T09:06:48.087533Z","iopub.execute_input":"2023-03-21T09:06:48.088549Z","iopub.status.idle":"2023-03-21T09:06:48.449923Z","shell.execute_reply.started":"2023-03-21T09:06:48.088501Z","shell.execute_reply":"2023-03-21T09:06:48.448662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# previewing a random sample\nx = df.sample(1)\npath = x['path'].values[0]\ncls = x['class'].values[0]\nprint('Bird species Id:', cls)\n\n#displaying raw audio signal \ny, sr = librosa.load(path, sr=None) #using the native sampling rate of the audio file\nprint('Sampling rate: ', sr)\nplt.plot(y);\nplt.title('Raw Waveform')\nplt.xlabel('Time (samples)')\nplt.ylabel('Amplitude')\n\n#Spectrogram and MFCC\"S ==> two most important feature extraction used for audio processing in deep learning\nplt.figure(figsize=(10, 5)) \ny, sr = librosa.load(path, sr=None)\n\nplt.subplot(1,2,1)\nS = librosa.feature.melspectrogram(y=y, sr=sr)\nS_DB = librosa.power_to_db(S, ref=np.max)\nlibrosa.display.specshow(S_DB, sr=sr, x_axis='time', y_axis='mel');\nplt.colorbar(format='%+2.0f dB');\nplt.title(\"Mel Spectrogram\")\n\nplt.subplot(1,2,2)\nmfccs = librosa.feature.mfcc(y=y, sr=sr, n_mfcc=40) #extracting 40-dimensional mfccs featues\nlibrosa.display.specshow(mfccs, x_axis='time')\nplt.title(\"MFCC\")\nplt.ylabel(\"MFCC coefficients\")\nplt.colorbar()\nplt.show()\n\n#palyer for listening the audio\nAudio(path)\n\n#Inference:\n#dataset is not quite clean it has a lot of background noise, but still bird calls are distinct enough to note it to some extend\n#each audio recordings is of different lenght","metadata":{"execution":{"iopub.status.busy":"2023-03-21T10:28:45.661672Z","iopub.execute_input":"2023-03-21T10:28:45.662203Z","iopub.status.idle":"2023-03-21T10:28:46.859980Z","shell.execute_reply.started":"2023-03-21T10:28:45.662159Z","shell.execute_reply":"2023-03-21T10:28:46.858602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ref: https://stackoverflow.com/a/45276885/12828621\n\n#NOTE: without mutliprocessing this cells takes more than 1 hour to run, \n#but with mutltiprocesing it runs in less than 5 minutes\n\n# plot the length of audio recordings as a distribution\n# list of audio file paths\naudio_files = df['path'].tolist()\n\n# get the duration of each file in seconds (using multiprocessing) (we have in kaggle 4cores)\ndef calculate_audio_length(filepath):\n    y, sr = librosa.load(filepath, sr=None)\n    length = librosa.get_duration(y=y, sr=sr)\n    return length\n\nn_cores = multiprocessing.cpu_count()\nprint('Available cores: ', n_cores)\nwith multiprocessing.Pool(n_cores) as p:\n    duration = list(tqdm.tqdm(p.imap(calculate_audio_length, audio_files), total=len(audio_files)))\n\n\n#plot\nsns.kdeplot(x=duration)\npercentile = np.percentile(duration, 99)\nprint('99 % of audio recordings is of length: ', round(percentile), ' seconds')\nplt.axvline(x=percentile, color='r', linestyle='--')\nplt.xlabel(\"Duration (s)\")\nplt.ylabel(\"Frequency\")\nplt.title(f\"Distribution of Audio Recording Durations.\\n 99% of audio recordings is of length: {round(percentile)} seconds\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T10:22:53.548466Z","iopub.execute_input":"2023-03-21T10:22:53.550008Z","iopub.status.idle":"2023-03-21T10:27:50.019807Z","shell.execute_reply.started":"2023-03-21T10:22:53.549940Z","shell.execute_reply":"2023-03-21T10:27:50.018058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}