{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        (os.path.join(dirname, filename))\n\nimport librosa","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-18T18:22:48.789817Z","iopub.execute_input":"2023-08-18T18:22:48.790342Z","iopub.status.idle":"2023-08-18T18:24:04.722080Z","shell.execute_reply.started":"2023-08-18T18:22:48.790292Z","shell.execute_reply":"2023-08-18T18:24:04.721028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This notebook is the memo to myself about various audio features and how they can be used to train models. I had prepared data which I am using in this notebook, I will remove some rows as it is a huge dataset","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/bird-clef-mfccs/MFCCs_For_Birds.csv\") #This is the MFCCs data which I prepared from train_audio data","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:24:04.723841Z","iopub.execute_input":"2023-08-18T18:24:04.724828Z","iopub.status.idle":"2023-08-18T18:25:30.617442Z","shell.execute_reply.started":"2023-08-18T18:24:04.724790Z","shell.execute_reply":"2023-08-18T18:25:30.616439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:30.618744Z","iopub.execute_input":"2023-08-18T18:25:30.619057Z","iopub.status.idle":"2023-08-18T18:25:30.644749Z","shell.execute_reply.started":"2023-08-18T18:25:30.619029Z","shell.execute_reply":"2023-08-18T18:25:30.643466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Initial rows and columns in the dataset","metadata":{}},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:30.647455Z","iopub.execute_input":"2023-08-18T18:25:30.647835Z","iopub.status.idle":"2023-08-18T18:25:30.655947Z","shell.execute_reply.started":"2023-08-18T18:25:30.647803Z","shell.execute_reply":"2023-08-18T18:25:30.654606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Categorically seeing which bird category is most prominent in the dataset","metadata":{}},{"cell_type":"code","source":"val = data.groupby('labels').count()\nval = val.rename(columns = {'Unnamed: 0':'count'})\nval.drop('mfcc', axis ='columns', inplace = True)\nval = val.sort_values(by =['count'], ascending = False)\nval.head(20)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:30.657338Z","iopub.execute_input":"2023-08-18T18:25:30.657721Z","iopub.status.idle":"2023-08-18T18:25:30.715718Z","shell.execute_reply.started":"2023-08-18T18:25:30.657691Z","shell.execute_reply":"2023-08-18T18:25:30.714585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I will be limiting samples of bird categories which are more than 1000 as it is imbalancing the datset, similarly I am removing the lower frequencies of categories from the dataset.","metadata":{}},{"cell_type":"code","source":"remove_tail = val.tail(50).index\nreduce_to_1000 = val.head(12).index\ndata.drop('Unnamed: 0', axis ='columns', inplace = True)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:30.717138Z","iopub.execute_input":"2023-08-18T18:25:30.718080Z","iopub.status.idle":"2023-08-18T18:25:30.728112Z","shell.execute_reply.started":"2023-08-18T18:25:30.718046Z","shell.execute_reply":"2023-08-18T18:25:30.726852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for x in reduce_to_1000:\n    lst = data.loc[data['labels'] == x].index.tolist()\n    data.drop(index = lst[1000:],inplace = True)\n    \nfor x in remove_tail:\n    lst = data.loc[data['labels'] == x].index.tolist()\n    data.drop(index = lst[:],inplace = True)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:30.729992Z","iopub.execute_input":"2023-08-18T18:25:30.730781Z","iopub.status.idle":"2023-08-18T18:25:31.762898Z","shell.execute_reply.started":"2023-08-18T18:25:30.730742Z","shell.execute_reply":"2023-08-18T18:25:31.761804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:31.764626Z","iopub.execute_input":"2023-08-18T18:25:31.765313Z","iopub.status.idle":"2023-08-18T18:25:31.772548Z","shell.execute_reply.started":"2023-08-18T18:25:31.765276Z","shell.execute_reply":"2023-08-18T18:25:31.771433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Audio Feature Exploration","metadata":{}},{"cell_type":"code","source":"audio_path = \"/kaggle/input/birdclef-2023/train_audio/abethr1/XC128013.ogg\" #Sample file\nfile, sample_rate = librosa.load(audio_path) ","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:31.773723Z","iopub.execute_input":"2023-08-18T18:25:31.774739Z","iopub.status.idle":"2023-08-18T18:25:44.970804Z","shell.execute_reply.started":"2023-08-18T18:25:31.774703Z","shell.execute_reply":"2023-08-18T18:25:44.969589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import IPython.display as ipd\nipd.Audio(audio_path)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:44.975713Z","iopub.execute_input":"2023-08-18T18:25:44.976354Z","iopub.status.idle":"2023-08-18T18:25:45.012526Z","shell.execute_reply.started":"2023-08-18T18:25:44.976318Z","shell.execute_reply":"2023-08-18T18:25:45.011549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Waveform depicts the change in the amplitude of the singal over course of time","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport librosa.display\nplt.figure(figsize=(14, 5))\nlibrosa.display.waveshow(file, sr=sample_rate)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:45.013548Z","iopub.execute_input":"2023-08-18T18:25:45.013914Z","iopub.status.idle":"2023-08-18T18:25:45.590220Z","shell.execute_reply.started":"2023-08-18T18:25:45.013885Z","shell.execute_reply":"2023-08-18T18:25:45.588839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Short time fourier transform is used to convert the time-domain amplitude signal to frequency domain signal, which results in the spectogram. Spectogram depicts changes in the frequency domain over time, and amplitude is usually represented in the third dimension.","metadata":{}},{"cell_type":"code","source":"#display Spectrogram\nX = librosa.stft(file)\nXdb = librosa.amplitude_to_db(abs(X))\nplt.figure(figsize=(14, 5))\nlibrosa.display.specshow(Xdb, sr=sample_rate, x_axis='time', y_axis='hz') \nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:45.591757Z","iopub.execute_input":"2023-08-18T18:25:45.592171Z","iopub.status.idle":"2023-08-18T18:25:48.427922Z","shell.execute_reply.started":"2023-08-18T18:25:45.592128Z","shell.execute_reply":"2023-08-18T18:25:48.426629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As can be seen from the above spectrogram, red streaks in between shows the sound of the bird which is roughly present at 2sec, 12 sec, 20 sec.... of the audio clip, this can also be confirmed by the waveform.","metadata":{}},{"cell_type":"code","source":"len(file)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:48.429778Z","iopub.execute_input":"2023-08-18T18:25:48.430466Z","iopub.status.idle":"2023-08-18T18:25:48.436986Z","shell.execute_reply.started":"2023-08-18T18:25:48.430429Z","shell.execute_reply":"2023-08-18T18:25:48.435760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Following code zooms in the specific section of the file, for example at 32000 we can see sharp changes in the following graph","metadata":{}},{"cell_type":"code","source":"n0 = 10000\nn1 = 50000\nplt.figure(figsize=(14, 5))\nplt.plot(file[n0:n1])\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:31:36.155468Z","iopub.execute_input":"2023-08-18T18:31:36.155933Z","iopub.status.idle":"2023-08-18T18:31:36.631300Z","shell.execute_reply.started":"2023-08-18T18:31:36.155895Z","shell.execute_reply":"2023-08-18T18:31:36.629474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ZCR(zero crossing rate) is how many times signal changes its sign, following ar e the applications of ZCR(this is from internet):\n\nThe ZCR has several applications and significance in audio signal processing:\n\nPitch Estimation: When a waveform has a higher rate of zero crossings, it might correspond to higher-pitched sounds, while lower rates could correspond to lower-pitched sounds.\n\nFeature Extraction: ZCR can provide valuable information about the rhythmic and percussive characteristics of an audio signal. For instance, it can help differentiate between speech and non-speech segments, or identify whether a signal contains silence or active sound.\n\nBoundary Detection: The ZCR can help in segmenting audio signals into meaningful regions, such as identifying where individual syllables or words occur in speech signals, or detecting note boundaries in music.\n\nNoise Detection and Reduction: Sudden changes in the zero crossing rate can indicate the presence of abrupt changes in the audio signal, which might be associated with noise or interference. This information can be used for noise detection and reduction in audio processing applications.\n\nEmotion Recognition: Some studies suggest that certain emotional states in speech might be associated with variations in the zero crossing rate. For instance, high emotional arousal might result in changes in speech rate and, consequently, the zero crossing rate.\n\nAudio Effects and Manipulation: The zero crossing rate can also be used creatively for audio effects and manipulation. For example, it can be employed to control effects like tremolo or gating in music production.","metadata":{}},{"cell_type":"code","source":"zero_crossings = librosa.zero_crossings(file[n0:n1], pad=False)\nprint(sum(zero_crossings))","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:33:00.500506Z","iopub.execute_input":"2023-08-18T18:33:00.501019Z","iopub.status.idle":"2023-08-18T18:33:00.521505Z","shell.execute_reply.started":"2023-08-18T18:33:00.500981Z","shell.execute_reply":"2023-08-18T18:33:00.520172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Spectral Centroid means the center of the mass in the audio signal. It has following applications:\n\nTimbral Characteristics: The spectral centroid can be thought of as a measure of the \"brightness\" or \"brightness center\" of a sound. Sounds with higher spectral centroids are perceived as brighter or having more high-frequency content, while sounds with lower centroids are perceived as darker or having more low-frequency content.\n\nInstrument Classification and Sound Analysis: Different instruments and sounds tend to have distinct spectral centroid ranges, allowing for discrimination between them based on their tonal characteristics.\n\nMusic Production and Equalization:  By adjusting the balance of frequencies around the spectral centroid, we can emphasize or de-emphasize specific tonal qualities. For instance, boosting the spectral centroid can make a sound brighter, while attenuating it can make it sound darker.\n\nAudio Effects: The spectral centroid can also influence the behavior of various audio effects.\n\nContent-Based Retrieval: In audio databases or content-based retrieval systems, the spectral centroid can be used as a feature for searching and retrieving sounds based on their tonal characteristics, helping users find similar sounds to those they input.","metadata":{}},{"cell_type":"code","source":"import sklearn\nspectral_centroids = librosa.feature.spectral_centroid(y = file, sr=sample_rate)[0]\nframes = range(len(spectral_centroids))\nt = librosa.frames_to_time(frames)\nlibrosa.display.waveshow(file, sr=sample_rate, alpha=0.4)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:48.780092Z","iopub.execute_input":"2023-08-18T18:25:48.780600Z","iopub.status.idle":"2023-08-18T18:25:51.203595Z","shell.execute_reply.started":"2023-08-18T18:25:48.780554Z","shell.execute_reply":"2023-08-18T18:25:51.202222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Spectral rolloff is the frequency below which a specified percentage of the total spectral energies.","metadata":{}},{"cell_type":"code","source":"spectral_rolloff = librosa.feature.spectral_rolloff(y = file, sr=sample_rate)[0]\nlibrosa.display.waveshow(file, sr=sample_rate, alpha=0.4)","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:51.205113Z","iopub.execute_input":"2023-08-18T18:25:51.205493Z","iopub.status.idle":"2023-08-18T18:25:51.858143Z","shell.execute_reply.started":"2023-08-18T18:25:51.205461Z","shell.execute_reply":"2023-08-18T18:25:51.856966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Saving the reduced dataset to a .csv file, I will be using this data for deep learning and I will try to use transformers on this dataset. I am only using MFCCs features per audio clip for model training","metadata":{}},{"cell_type":"code","source":"data.to_csv(\"Final_reduced_data.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-08-18T18:25:51.859953Z","iopub.execute_input":"2023-08-18T18:25:51.860295Z","iopub.status.idle":"2023-08-18T18:26:37.790076Z","shell.execute_reply.started":"2023-08-18T18:25:51.860266Z","shell.execute_reply":"2023-08-18T18:26:37.788010Z"},"trusted":true},"execution_count":null,"outputs":[]}]}