{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nfrom scipy.io import wavfile\nimport librosa\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport random\nrandom.seed(111)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"This is my quick implementation of SpecAugment paper [here](https://arxiv.org/abs/1904.08779), without time warping. It works regardless of PyTorch or Tensorflow.\n\nYou set percentage of frames to mask so should work with long and short segments.\n\nLet's test it:"},{"metadata":{"trusted":true},"cell_type":"code","source":"def spec_augment(spec: np.ndarray, num_mask=2, \n                 freq_masking_max_percentage=0.15, time_masking_max_percentage=0.3):\n\n    spec = spec.copy()\n    for i in range(num_mask):\n        all_frames_num, all_freqs_num = spec.shape\n        freq_percentage = random.uniform(0.0, freq_masking_max_percentage)\n        \n        num_freqs_to_mask = int(freq_percentage * all_freqs_num)\n        f0 = np.random.uniform(low=0.0, high=all_freqs_num - num_freqs_to_mask)\n        f0 = int(f0)\n        spec[:, f0:f0 + num_freqs_to_mask] = 0\n\n        time_percentage = random.uniform(0.0, time_masking_max_percentage)\n        \n        num_frames_to_mask = int(time_percentage * all_frames_num)\n        t0 = np.random.uniform(low=0.0, high=all_frames_num - num_frames_to_mask)\n        t0 = int(t0)\n        spec[t0:t0 + num_frames_to_mask, :] = 0\n    \n    return spec\n    ","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"audio_path = os.path.join(\"../input/train_curated/d7d25898.wav\")\nsr, audio = wavfile.read(audio_path)\n\nx = librosa.feature.melspectrogram(y=audio.astype(float), sr=sr, S=None, n_fft=512, hop_length=256, n_mels=40).T\nx = librosa.power_to_db(x, ref=np.max)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure()\nplt.imshow(spec_augment(x),aspect= 'auto')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure()\nplt.imshow(spec_augment(x),aspect= 'auto')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.figure()\nplt.imshow(spec_augment(x),aspect= 'auto')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}