{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import librosa\nimport soundfile as sf\nimport scipy.signal as signal\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nfrom librosa import display as ld\nfrom matplotlib.patches import Rectangle","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"data_dir = '/kaggle/input/rfcx-species-audio-detection/'\n\ndf_t = pd.read_csv(data_dir+'train_tp.csv')\ndf_t[\"t_dif\"] = df_t[\"t_max\"] - df_t[\"t_min\"]\ndf_t[\"f_dif\"] = df_t[\"f_max\"] - df_t[\"f_min\"]\n\ndf_f = pd.read_csv(data_dir+'train_fp.csv')\ndf_f[\"t_dif\"] = df_f[\"t_max\"] - df_f[\"t_min\"]\ndf_f[\"f_dif\"] = df_f[\"f_max\"] - df_f[\"f_min\"]\n\ndf_t.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Get example"},{"metadata":{"trusted":true},"cell_type":"code","source":"example = df_t.iloc[0]\ndata, samplerate = sf.read(data_dir+'train/'+example.recording_id+'.flac')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Audio sound"},{"metadata":{"trusted":true},"cell_type":"code","source":"import IPython.display as ipd\nipd.Audio(data,rate=samplerate)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"times = np.linspace(0,len(data),len(data))/samplerate\nplt.figure(figsize=(30, 10))\nplt.plot(times,data)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# SPECTOGRAM"},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(30, 10))\nPxx, freqs, bins, im = plt.specgram(data, Fs=samplerate)\n\nplt.ylabel('Frequency [Hz]')\nplt.xlabel('Time [sec]')\n# Add the audio position\nax = plt.gca()\naudio_position = Rectangle((example['t_min'],example['f_min']),example['t_dif'],example['f_dif'],linewidth=3,edgecolor='g',facecolor='none')\nax.add_patch(audio_position)\n\nplt.colorbar()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# SFTF"},{"metadata":{"trusted":true},"cell_type":"code","source":"sftf = librosa.stft(data)\n\nplt.figure(figsize=(30, 10))\nld.specshow(sftf, sr=samplerate, x_axis='time', y_axis='hz')\n\n# Add the audio position\nax = plt.gca()\naudio_position = Rectangle((example['t_min'],example['f_min']),example['t_dif'],example['f_dif'],linewidth=3,edgecolor='g',facecolor='none')\nax.add_patch(audio_position)\n\nplt.colorbar()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# SFTF (amplitude_to_db)"},{"metadata":{"trusted":true},"cell_type":"code","source":"sftf = librosa.stft(data)\nsftf_xdb = librosa.amplitude_to_db(abs(sftf))\n\nplt.figure(figsize=(30, 10))\nld.specshow(sftf_xdb, sr=samplerate, x_axis='time', y_axis='hz')\n\n# Add the audio position\nax = plt.gca()\naudio_position = Rectangle((example['t_min'],example['f_min']),example['t_dif'],example['f_dif'],linewidth=3,edgecolor='g',facecolor='none')\nax.add_patch(audio_position)\n\nplt.colorbar()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# MEL SPECTOGRAM"},{"metadata":{"trusted":true},"cell_type":"code","source":"melspec = librosa.feature.melspectrogram(data, sr=samplerate)\n\nplt.figure(figsize=(30, 10))\nld.specshow(melspec, sr=samplerate, x_axis='time', y_axis='hz')\n\n# Add the audio position\nax = plt.gca()\naudio_position = Rectangle((example['t_min'],example['f_min']),example['t_dif'],example['f_dif'],linewidth=3,edgecolor='g',facecolor='none')\nax.add_patch(audio_position)\nplt.colorbar()\n\nplt.show()\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# MEL SPECTOGRAM (amplitude_to_db)"},{"metadata":{"trusted":true},"cell_type":"code","source":"melspec = librosa.feature.melspectrogram(data, sr=samplerate)\nmelspec_xdb = librosa.amplitude_to_db(abs(melspec))\n\nplt.figure(figsize=(30, 10))\nld.specshow(melspec_xdb, sr=samplerate, x_axis='time', y_axis='hz')\n\n# Add the audio position\nax = plt.gca()\naudio_position = Rectangle((example['t_min'],example['f_min']),example['t_dif'],example['f_dif'],linewidth=3,edgecolor='g',facecolor='none')\nax.add_patch(audio_position)\nplt.colorbar()\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# MFCC"},{"metadata":{"trusted":true},"cell_type":"code","source":"mfcc = librosa.feature.mfcc(data, sr=samplerate)\n\nplt.figure(figsize=(30, 10))\nld.specshow(mfcc, sr=samplerate, x_axis='time', y_axis='hz')\n\n# Add the audio position\nax = plt.gca()\naudio_position = Rectangle((example['t_min'],example['f_min']),example['t_dif'],example['f_dif'],linewidth=3,edgecolor='g',facecolor='none')\nax.add_patch(audio_position)\nplt.colorbar()\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# MFCC (amplitude_to_db)"},{"metadata":{"trusted":true},"cell_type":"code","source":"mfcc = librosa.feature.mfcc(data, sr=samplerate)\nmfcc_xdb = librosa.amplitude_to_db(abs(mfcc))\n\nplt.figure(figsize=(30, 10))\nld.specshow(mfcc_xdb, sr=samplerate, x_axis='time', y_axis='hz')\n\n# Add the audio position\nax = plt.gca()\naudio_position = Rectangle((example['t_min'],example['f_min']),example['t_dif'],example['f_dif'],linewidth=3,edgecolor='g',facecolor='none')\nax.add_patch(audio_position)\nplt.colorbar()\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# CHROMA_STFT"},{"metadata":{"trusted":true},"cell_type":"code","source":"chroma = librosa.feature.chroma_stft(data, sr=samplerate)\n\nplt.figure(figsize=(30, 10))\nld.specshow(chroma, sr=samplerate, x_axis='time', y_axis='hz')\n\n# Add the audio position\nax = plt.gca()\naudio_position = Rectangle((example['t_min'],example['f_min']),example['t_dif'],example['f_dif'],linewidth=3,edgecolor='g',facecolor='none')\nax.add_patch(audio_position)\nplt.colorbar()\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# CHROMA_STFT (amplitude_to_db)"},{"metadata":{"trusted":true},"cell_type":"code","source":"chroma = librosa.feature.chroma_stft(data, sr=samplerate)\nchroma_xdb = librosa.amplitude_to_db(abs(chroma))\n\nplt.figure(figsize=(30, 10))\nld.specshow(chroma_xdb, sr=samplerate, x_axis='time', y_axis='hz')\n\n# Add the audio position\nax = plt.gca()\naudio_position = Rectangle((example['t_min'],example['f_min']),example['t_dif'],example['f_dif'],linewidth=3,edgecolor='g',facecolor='none')\nax.add_patch(audio_position)\nplt.colorbar()\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}