{"cells":[{"metadata":{},"cell_type":"markdown","source":"<h2> 1. Importing required libraries and dataset </h2>"},{"metadata":{"trusted":true},"cell_type":"code","source":"! pip install librosa","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n%matplotlib inline\nfrom matplotlib import pyplot as plt\n\nimport IPython.display as ipd\n\nimport os\nimport os.path\nfrom os import path\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\nimport time\n\nfrom tqdm import tqdm\nfrom tqdm.notebook import tqdm\n\n\nimport librosa \nimport librosa.display\n#Librosa is a python package for music and audio analysis.\n\nimport tensorflow as tf","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf=pd.read_csv(r\"/kaggle/input/rfcx-species-audio-detection/train_tp.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(f\"total Species in Dataset : {len(traindf.recording_id)}\")\nprint(f\"total Recordings in Dataset : {len(traindf.recording_id.unique())}\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Creating an Index column as there is redundancy in Recording Id Column, May need a unique identifier"},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf['idx'] = range(1, len(traindf) + 1)\ntraindf.head(5)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<h2>2. Data Analysis </h2>"},{"metadata":{},"cell_type":"markdown","source":"#### Let's analyse the number of Species ID we have and their frequencies using count plot "},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(traindf['species_id'])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"#### Species_id is balanced, all the categories have sufficient number of entries\nLet's look into other features"},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.countplot(traindf['songtype_id'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp=traindf.loc[(traindf['songtype_id'] ==4)]\ntemp.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path.exists(\"/kaggle/input/rfcx-species-audio-detection/train/0099c367b.flac\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"## playing audio for songtype_id=4\nstart = time.perf_counter()\ndata1, sr1 = librosa.load('/kaggle/input/rfcx-species-audio-detection/train/0099c367b.flac')\nprint(\"Time taken - \" + str(time.perf_counter()-start))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ipd.Audio('/kaggle/input/rfcx-species-audio-detection/train/003bec244.flac')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"path='/kaggle/input/rfcx-species-audio-detection/train/003bec244.flac'\ny, sr = librosa.load(path)\nplt.figure(figsize=(20,5))\nlibrosa.display.waveplot(y, sr=sr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#display Spectrogram\npath='/kaggle/input/rfcx-species-audio-detection/train/003bec244.flac'\nx, sr = librosa.load(path)\nX = librosa.stft(x)\nXdb = librosa.amplitude_to_db(abs(X))\nplt.figure(figsize=(20, 5))\nlibrosa.display.specshow(Xdb, sr=sr, x_axis='time', y_axis='hz') \n#If to pring log of frequencies  \n#librosa.display.specshow(Xdb, sr=sr, x_axis='time', y_axis='log')\nplt.colorbar()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<h2>3. Data Preparation </h2>\n\n* Creating Train and Validation split\n* Triming the audio\n* Reading the audio files with labels"},{"metadata":{"trusted":true},"cell_type":"code","source":"# Getting list of all filenames and shuffling them\nIDX=np.random.permutation(traindf['idx'])\nnum_samples = len(IDX)\nprint('Number of total examples:', num_samples)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_idx = IDX[:850]\nval_idx = IDX[850: 1216]\nprint('Training set size', len(train_idx))\nprint('Validation set size', len(val_idx))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def trim_audio(IDX):\n    recording_id=traindf.loc[traindf['idx'] == IDX,'recording_id'].iloc[0]\n    t_min=traindf.loc[traindf['idx'] == IDX,'t_min'].iloc[0]\n    t_max=traindf.loc[traindf['idx'] == IDX,'t_max'].iloc[0]\n    label=traindf.loc[traindf['idx'] == IDX,'species_id'].iloc[0]\n    #print(f\"recording_id == {recording_id} & t_min = {t_min} & t_max = {t_max}\")\n    path='/kaggle/input/rfcx-species-audio-detection/train/'+recording_id+'.flac'\n    #print(f\"Path == {path}\")\n    y, sr = librosa.load(path)\n    time_start = t_min*sr\n    time_end = t_max*sr\n    # Positioning sound slice\n    # taking 5 seconds around center of the start and end time and cropping it accordingly\n    effective_length=10*sr\n    center = np.round((time_start + time_end) / 2)\n    beginning = center - effective_length / 2\n    if beginning < 0:\n        beginning = 0\n    beginning = np.random.randint(beginning , center)\n    ending = beginning + effective_length\n    if ending > len(y):\n        ending = len(y)\n    beginning = ending - effective_length\n    y = y[beginning:ending].astype(np.float32)\n    \n    beginning_time = beginning / sr\n    ending_time = ending / sr\n    \n    #query_string = f\"recording_id == {recording_id} & \"\n    #query_string += f\"t_min = {beginning_time} & t_max = {ending_time}\"\n    #print(query_string)\n    #y=tf.squeeze(y, axis=-1)\n    return y,label","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Creating train and validation datasets"},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf_new = pd.DataFrame(columns = ['audio', 'label'])\nvaldf_new=pd.DataFrame(columns = ['audio', 'label'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for index in tqdm(train_idx):\n    d,l=trim_audio(index)\n    traindf_new = traindf_new.append({'audio' : d, 'label' : l},ignore_index = True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf_new.head(10)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def convert_tolist(x):\n    return x.tolist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf_new1=traindf_new\nconvert_tolist(traindf_new.iloc[5].audio)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf_new1.audio=traindf_new1.audio.apply(convert_tolist)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf_new1.to_csv('TrainWaveForm.csv', index = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"traindf_new.iloc[5].audio","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ipd.Audio(traindf_new.iloc[5].audio, rate=16000)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for index in tqdm(val_idx):\n    d,l=trim_audio(index)\n    valdf_new = valdf_new.append({'audio' : d, 'label' : l},ignore_index = True) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valdf_new1=valdf_new\nvaldf_new1.audio=valdf_new1.audio.apply(convert_tolist)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"valdf_new1.to_csv('ValWaveForm.csv', index = False)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"<center><h3>IN PROGRESS</h3></center>"},{"metadata":{},"cell_type":"markdown","source":"### Helpful Resources\n<hr/>\n\n* https://www.youtube.com/watch?v=iCwMQJnKk2c&list=PL-wATfeyAMNqIee7cH3q1bh4QJFAaeNv0\n* https://www.kaggle.com/c/rfcx-species-audio-detection/discussion/200922\n* https://medium.com/@patrickbfuller/librosa-a-python-audio-libary-60014eeaccfb \n* https://towardsdatascience.com/extract-features-of-music-75a3f9bc265d\n* https://www.kdnuggets.com/2020/02/audio-data-analysis-deep-learning-python-part-1.html"},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}