{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport librosa\nimport librosa.display\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"audio_path = '../input/birdsong-recognition/train_audio/aldfly/XC134874.mp3'\nx , sr = librosa.load(audio_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(16, 4))\nlibrosa.display.waveplot(x, sr=sr)\nplt.title('Slower Version $X_1$')\nplt.tight_layout()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"S = librosa.feature.melspectrogram(x, sr=sr, n_mels=128)\n\n# Convert to log scale (dB). We'll use the peak power (max) as reference.\nlog_S = librosa.power_to_db(S, ref=np.max)\n\n# Make a new figure\nplt.figure(figsize=(12,4))\n\n# Display the spectrogram on a mel scale\n# sample rate and hop length parameters are used to render the time axis\nlibrosa.display.specshow(log_S, sr=sr, x_axis='time', y_axis='mel')\n\n# Put a descriptive title on the plot\nplt.title('mel power spectrogram')\n\n# draw a color bar\nplt.colorbar(format='%+02.0f dB')\n\n# Make the figure layout compact\nplt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import IPython.display as ipd\nipd.Audio(audio_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_harmonic, y_percussive = librosa.effects.hpss(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"\nC = librosa.feature.chroma_cqt(y=y_harmonic, sr=sr, bins_per_octave=36)\n\n# Make a new figure\nplt.figure(figsize=(12,4))\n\n# Display the chromagram: the energy in each chromatic pitch class as a function of time\n# To make sure that the colors span the full range of chroma values, set vmin and vmax\nlibrosa.display.specshow(C, sr=sr, x_axis='time', y_axis='chroma', vmin=0, vmax=1)\n\nplt.title('Chromagram')\nplt.colorbar()\n\nplt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ipd.Audio(data=y_harmonic, rate=sr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ipd.Audio(data=y_percussive, rate=sr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_shift = librosa.effects.pitch_shift(x, sr, 7)\n\nipd.Audio(data=y_shift, rate=sr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"D = librosa.stft(x)\n\n# Separate the magnitude and phase\nS, phase = librosa.magphase(D)\n\n# Decompose by nmf\ncomponents, activations = librosa.decompose.decompose(S, n_components=8, sort=True)\n\n\nplt.figure(figsize=(12,4))\n\nplt.subplot(1,2,1)\nlibrosa.display.specshow(librosa.amplitude_to_db(np.abs(components), ref=np.max), y_axis='log')\nplt.xlabel('Component')\nplt.ylabel('Frequency')\nplt.title('Components')\n\nplt.subplot(1,2,2)\nlibrosa.display.specshow(activations, x_axis='time')\nplt.xlabel('Time')\nplt.ylabel('Component')\nplt.title('Activations')\n\nplt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# we isolate just last (highest) component?\nk = -1\n\n# Reconstruct a spectrogram by the outer product of component k and its activation\nD_k = np.multiply.outer(components[:, k], activations[k])\n\n# invert the stft after putting the phase back in\ny_k = librosa.istft(D_k * phase)\n\n# And playback\nprint('Component #{}'.format(k))\n\nipd.Audio(data=y_k, rate=sr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def  apply_tunning(y):\n    '''Load audio, estimate tuning, apply pitch correction, and save.'''\n  \n\n    print('Separating harmonic component ... ')\n    y_harm = librosa.effects.harmonic(y)\n\n    print('Estimating tuning ... ')\n    # Just track the pitches associated with high magnitude\n    tuning = librosa.estimate_tuning(y=y_harm, sr=sr)\n\n    print('{:+0.2f} cents'.format(100 * tuning))\n    print('Applying pitch-correction of {:+0.2f} cents'.format(-100 * tuning))\n    y_tuned = librosa.effects.pitch_shift(y, sr, -tuning)\n    return  y_tuned\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_tuned = apply_tunning(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"y_tuned","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Next, we'll extract the top 13 Mel-frequency cepstral coefficients (MFCCs)\nS = librosa.feature.melspectrogram(x, sr=sr, n_mels=128)\n\n# Convert to log scale (dB). We'll use the peak power (max) as reference.\nlog_S = librosa.power_to_db(S, ref=np.max)\nmfcc        = librosa.feature.mfcc(S=log_S, n_mfcc=15)\n\n# Let's pad on the first and second deltas while we're at it\ndelta_mfcc  = librosa.feature.delta(mfcc)\ndelta2_mfcc = librosa.feature.delta(mfcc, order=2)\n\n# How do they look?  We'll show each in its own subplot\nplt.figure(figsize=(12, 6))\n\nplt.subplot(3,1,1)\nlibrosa.display.specshow(mfcc)\nplt.ylabel('MFCC')\nplt.colorbar()\n\nplt.subplot(3,1,2)\nlibrosa.display.specshow(delta_mfcc)\nplt.ylabel('MFCC-$\\Delta$')\nplt.colorbar()\n\nplt.subplot(3,1,3)\nlibrosa.display.specshow(delta2_mfcc, sr=sr, x_axis='time')\nplt.ylabel('MFCC-$\\Delta^2$')\nplt.colorbar()\n\nplt.tight_layout()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"S_full, phase = librosa.magphase(librosa.stft(x, hop_length=2048, window=np.ones))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"idx = slice(*librosa.time_to_frames([10, 15], hop_length=2048, sr=sr))\nplt.figure(figsize=(12, 4))\nlibrosa.display.specshow(librosa.power_to_db(S_full[:, idx]**2, ref=np.max),\n                         y_axis='log', x_axis='time', sr=sr)\nplt.colorbar()\nplt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"S_filter = librosa.decompose.nn_filter(S_full,\n                                       aggregate=np.median,\n                                       metric='cosine',\n                                       k=20,\n                                       width=int(librosa.time_to_frames(2, sr=sr)))\n\n# The output of the filter shouldn't be greater than the input if we assume signals are additive\nS_filter = np.minimum(S_full, S_filter)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Note: the margins need not be equal for foreground and background separation\nmargin_i, margin_v = 2, 10\npower = 2\n\nmask_i = librosa.util.softmask(S_filter,\n                               margin_i * (S_full - S_filter),\n                               power=power)\n\nmask_v = librosa.util.softmask(S_full - S_filter,\n                               margin_v * S_filter,\n                               power=power)\n\n# Using the masks we get a cleaner signal\n\nS_foreground = mask_v * S_full\nS_background = mask_i * S_full","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure(figsize=(12, 8))\nplt.subplot(2,1,1)\nlibrosa.display.specshow(librosa.power_to_db(S_background[:, idx]**2, ref=np.max),\n                         y_axis='log', sr=sr)\nplt.title('Background')\nplt.colorbar()\nplt.subplot(2,1,2)\nlibrosa.display.specshow(librosa.power_to_db(S_foreground[:, idx]**2, ref=np.max),\n                         y_axis='log', x_axis='time', sr=sr)\nplt.title('Foreground')\nplt.colorbar()\nplt.tight_layout()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ipd.Audio(data=librosa.istft(S_background * phase), rate=sr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ipd.Audio(data=librosa.istft(S_foreground * phase), rate=sr)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}