{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"- Reference\n    - [RFCX: Audio Data Augmentation(Japanese+English) by: Hidehisa Arai](https://www.kaggle.com/hidehisaarai1213/rfcx-audio-data-augmentation-japanese-english)\n    - [Bird 2022 EDA - [Twitch Live Stream] by: Rob Mulla](https://www.kaggle.com/robikscube/bird-2022-eda-twitch-live-stream)\n    - [BirdCLEF_2022_Starter by: DrCapa](https://www.kaggle.com/drcapa/birdclef-2022-starter)\n    - [🦜BirdCLEF: world🌎map with birds by: Jirka Borovec](https://www.kaggle.com/jirkaborovec/birdclef-world-map-with-birds)\n    - [🦜BirdCLEF: EDA 🔎 & more... by: Jirka Borovec](https://www.kaggle.com/jirkaborovec/birdclef-eda-more)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport json\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom glob import glob\nimport matplotlib.pyplot as plt\n# import japanize_matplotlib\nimport seaborn as sns\n%matplotlib inline\nplt.style.use('fivethirtyeight')\n\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\nfrom IPython.display import Audio\n\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:29:56.413290Z","iopub.execute_input":"2022-02-17T16:29:56.414034Z","iopub.status.idle":"2022-02-17T16:29:56.421734Z","shell.execute_reply.started":"2022-02-17T16:29:56.413986Z","shell.execute_reply":"2022-02-17T16:29:56.421164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_dir = Path('../input/birdclef-2022')\nos.listdir(base_dir)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:29:56.423035Z","iopub.execute_input":"2022-02-17T16:29:56.423572Z","iopub.status.idle":"2022-02-17T16:29:56.442087Z","shell.execute_reply.started":"2022-02-17T16:29:56.423522Z","shell.execute_reply":"2022-02-17T16:29:56.441462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Data","metadata":{}},{"cell_type":"code","source":"train_meta = pd.read_csv(base_dir/'train_metadata.csv')\ntest_data = pd.read_csv(base_dir/'test.csv')\nebird_data = pd.read_csv(base_dir/'eBird_Taxonomy_v2021.csv')\nsamp_subm = pd.read_csv(base_dir/'sample_submission.csv')\n\nwith open(base_dir/'scored_birds.json') as f:\n    scored_birds = json.load(f)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:29:56.443248Z","iopub.execute_input":"2022-02-17T16:29:56.444749Z","iopub.status.idle":"2022-02-17T16:29:56.575939Z","shell.execute_reply.started":"2022-02-17T16:29:56.444709Z","shell.execute_reply":"2022-02-17T16:29:56.574994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_meta.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:29:56.577155Z","iopub.execute_input":"2022-02-17T16:29:56.577837Z","iopub.status.idle":"2022-02-17T16:29:56.593581Z","shell.execute_reply.started":"2022-02-17T16:29:56.577806Z","shell.execute_reply":"2022-02-17T16:29:56.592875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_meta.common_name.value_counts(ascending=True).plot.barh(figsize=(3, 24), grid=True) ","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:29:56.595202Z","iopub.execute_input":"2022-02-17T16:29:56.595535Z","iopub.status.idle":"2022-02-17T16:29:59.328932Z","shell.execute_reply.started":"2022-02-17T16:29:56.595507Z","shell.execute_reply":"2022-02-17T16:29:59.328277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_meta.primary_label.value_counts(ascending=True).plot.barh(figsize=(3, 24), grid=True) \n","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:29:59.330039Z","iopub.execute_input":"2022-02-17T16:29:59.330384Z","iopub.status.idle":"2022-02-17T16:30:01.269394Z","shell.execute_reply.started":"2022-02-17T16:29:59.330348Z","shell.execute_reply":"2022-02-17T16:30:01.268446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_count(feature, title, df, size=1):\n    '''クラス/特徴量をプロットする\n    Pram:\n        feature : 分析するカラム\n        title : グラフタイトル\n        df : プロットするデータフレーム\n        size : デフォルト 1.\n    '''\n    f, ax = plt.subplots(1,1, figsize=(4*size,4))\n    total = float(len(df))\n    # 最大10カラムをヒストグラムで表示\n#     g = sns.countplot(x = df[feature], order = df[feature].value_counts().index[:20], palette='Set3')\n    g = sns.countplot(y = df[feature], order = df[feature].value_counts().index[:10], palette='Set3')\n\n    g.set_title('Number and percentage of {}'.format(title))\n#     if(size > 2):\n        # サイズ2以上の時、行名を90°回転し、表示\n#         plt.xticks(rotation=90, size=8)\n    # データ比率の表示\n    for p in ax.patches:\n        height = p.get_height()\n        width = p.get_width()\n#         ax.text(p.get_x()+p.get_width()/2.,\n#                 height + 3,\n#                 '{:1.2f}%'.format(100*height/total),\n#                 ha='center')\n        ax.text(width + 500,\n                p.get_y()+p.get_height()+.025/2.,\n                '{:1.2f}%'.format(100*width/total),\n                ha='right') \n    plt.tight_layout()\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:01.270633Z","iopub.execute_input":"2022-02-17T16:30:01.270841Z","iopub.status.idle":"2022-02-17T16:30:01.278607Z","shell.execute_reply.started":"2022-02-17T16:30:01.270816Z","shell.execute_reply":"2022-02-17T16:30:01.277726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rating の比率を確認する\nplot_count(feature='rating', title='rating', df=train_meta, size=2)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:01.280005Z","iopub.execute_input":"2022-02-17T16:30:01.280279Z","iopub.status.idle":"2022-02-17T16:30:01.563163Z","shell.execute_reply.started":"2022-02-17T16:30:01.280249Z","shell.execute_reply":"2022-02-17T16:30:01.562525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# secondary_labels の比率を確認する\nplot_count(feature='secondary_labels', title='secondary_labels', df=train_meta, size=3)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:01.564541Z","iopub.execute_input":"2022-02-17T16:30:01.564997Z","iopub.status.idle":"2022-02-17T16:30:01.850996Z","shell.execute_reply.started":"2022-02-17T16:30:01.564964Z","shell.execute_reply":"2022-02-17T16:30:01.850349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from wordcloud import WordCloud, STOPWORDS, ImageColorGenerator\n\nstopwords = set(STOPWORDS)\nwordcloud = WordCloud(width = 5000, height = 4000,\n                      background_color ='white',\n                      stopwords = stopwords,\n                      min_font_size = 10).generate(' '.join(train_meta.secondary_labels))\n\nprint(wordcloud)\nfig = plt.figure(1)\nplt.figure(figsize=(10,5))\nplt.imshow(wordcloud, interpolation='bilinear')\nplt.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:01.852312Z","iopub.execute_input":"2022-02-17T16:30:01.852732Z","iopub.status.idle":"2022-02-17T16:30:31.009525Z","shell.execute_reply.started":"2022-02-17T16:30:01.852701Z","shell.execute_reply":"2022-02-17T16:30:31.008461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q GeoPandas","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:31.010754Z","iopub.execute_input":"2022-02-17T16:30:31.010985Z","iopub.status.idle":"2022-02-17T16:30:38.193002Z","shell.execute_reply.started":"2022-02-17T16:30:31.010956Z","shell.execute_reply":"2022-02-17T16:30:38.191949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import geopandas as gpd\n\n# The easiest way to plot data from Pandas on a world map\n# https://www.kaggle.com/jirkaborovec/birdclef-world-map-with-birds\n\n# initialize an axis\nfig, ax = plt.subplots(figsize=(16,12))\n# plot map on axis\ncountries = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres'))\ncountries.plot(color='lightgrey', ax=ax)\n\n# plot points\n# latitude=緯度, longitude=経度\ntrain_meta.plot.scatter(x='longitude', y='latitude', s=5, ax=ax) # , c='brightness'\n\nax.grid(b=True, alpha=0.25)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:38.195061Z","iopub.execute_input":"2022-02-17T16:30:38.195347Z","iopub.status.idle":"2022-02-17T16:30:38.675196Z","shell.execute_reply.started":"2022-02-17T16:30:38.195317Z","shell.execute_reply":"2022-02-17T16:30:38.674356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# initialize an axis\nfig, ax = plt.subplots(figsize=(24,18))\n# plot map on axis\ncountries = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres'))\ncountries.plot(color='lightgrey', ax=ax)\n\n# plot points\ncmap = plt.cm.get_cmap('jet')\nbirds = len(train_meta['primary_label'].unique())\nfor i, (bird, dfg) in enumerate(train_meta.groupby('primary_label')):\n    dfg.longitude = np.around(dfg.longitude, 1)\n    dfg.latitude = np.around(dfg.latitude, 1)\n    dfgg = dfg.groupby(['longitude', 'latitude']).size().reset_index(name='counts')\n    dfgg.plot(x='longitude', y='latitude', kind='scatter', c=cmap(float(i) / birds), s=dfgg['counts'] * 5, ax=ax, label=bird, alpha=0.5)\n\nax.legend(loc='upper center', bbox_to_anchor=(0.5, 1.25), ncol=15, fancybox=True, shadow=True)\n\n# get axes limits\nx_lo, x_up = ax.get_xlim()\ny_lo, y_up = ax.get_ylim()\n# add minor ticks with a specified sapcing (deg)\ndeg = 5\n# add grid\nax.set_xticks(np.arange(np.ceil(x_lo), np.ceil(x_up), deg), minor=True)\nax.set_yticks(np.arange(np.ceil(y_lo), np.ceil(y_up), deg), minor=True)\nax.grid(b=True, which='minor', alpha=0.25)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:38.678369Z","iopub.execute_input":"2022-02-17T16:30:38.678600Z","iopub.status.idle":"2022-02-17T16:30:53.232774Z","shell.execute_reply.started":"2022-02-17T16:30:38.678572Z","shell.execute_reply":"2022-02-17T16:30:53.231930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load Audio Data","metadata":{}},{"cell_type":"code","source":"# afrsil1 の音声リストの取得\nafrsil1_ogg_files = list(glob(f'{base_dir}/train_audio/afrsil1/*.ogg'))\nafrsil1_ogg_files[:5], len(afrsil1_ogg_files)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:53.234135Z","iopub.execute_input":"2022-02-17T16:30:53.234369Z","iopub.status.idle":"2022-02-17T16:30:53.243419Z","shell.execute_reply.started":"2022-02-17T16:30:53.234342Z","shell.execute_reply":"2022-02-17T16:30:53.242480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ファイル最初の先頭から10秒の区間を読み込み\ny, sr = librosa.load(afrsil1_ogg_files[0], duration=10)\n# data\nprint(y)\nprint()\n# rate\nprint(sr)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:53.244733Z","iopub.execute_input":"2022-02-17T16:30:53.244976Z","iopub.status.idle":"2022-02-17T16:30:53.572600Z","shell.execute_reply.started":"2022-02-17T16:30:53.244948Z","shell.execute_reply":"2022-02-17T16:30:53.572018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Audio(y, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:53.573822Z","iopub.execute_input":"2022-02-17T16:30:53.574029Z","iopub.status.idle":"2022-02-17T16:30:53.591669Z","shell.execute_reply.started":"2022-02-17T16:30:53.574004Z","shell.execute_reply":"2022-02-17T16:30:53.590944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 音声波形の表示\nx = range(len(y))\nplt.plot(x, y)\nplt.plot(x, y, color='blue')\nplt.legend(loc='upper center')\nplt.grid()\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:53.593036Z","iopub.execute_input":"2022-02-17T16:30:53.593521Z","iopub.status.idle":"2022-02-17T16:30:54.029331Z","shell.execute_reply.started":"2022-02-17T16:30:53.593477Z","shell.execute_reply":"2022-02-17T16:30:54.028524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# メル尺度のスペクトログラムの算出\nmel_spec = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=128)\n# スペクトログラム/クロマトグラム/ cqt /などを表示\nlibrosa.display.specshow(mel_spec, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:54.030697Z","iopub.execute_input":"2022-02-17T16:30:54.031087Z","iopub.status.idle":"2022-02-17T16:30:54.455521Z","shell.execute_reply.started":"2022-02-17T16:30:54.031050Z","shell.execute_reply":"2022-02-17T16:30:54.454812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# パワースペクトログラム（振幅の2乗）をデシベル（dB）単位に変換\n# ログスケールに変換\nmelspec = librosa.power_to_db(mel_spec)\n# スペクトログラム/クロマトグラム/ cqt /などを表示\nlibrosa.display.specshow(melspec, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:54.456462Z","iopub.execute_input":"2022-02-17T16:30:54.456773Z","iopub.status.idle":"2022-02-17T16:30:54.813595Z","shell.execute_reply.started":"2022-02-17T16:30:54.456746Z","shell.execute_reply":"2022-02-17T16:30:54.812736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Augmentation\n\n1. AddGaussianNoise\n2. GaussianNoiseSNR\n3. PinkNoiseSNR\n4. PitchShift\n5. TimeStretch\n6. TimeShift\n7. VolumeControl","metadata":{}},{"cell_type":"code","source":"class AudioTransform:\n    def __init__(self, always_apply=False, p=0.5):\n        self.always_apply = always_apply\n        self.p = p\n\n    def __call__(self, y: np.ndarray):\n        if self.always_apply:\n            return self.apply(y)\n        else:\n            if np.random.rand() < self.p:\n                return self.apply(y)\n            else:\n                return y\n\n    def apply(self, y: np.ndarray):\n        raise NotImplementedError\n\n\nclass Compose:\n    def __init__(self, transforms: list):\n        self.transforms = transforms\n\n    def __call__(self, y: np.ndarray):\n        for trns in self.transforms:\n            y = trns(y)\n        return y\n\n\nclass OneOf:\n    def __init__(self, transforms: list):\n        self.transforms = transforms\n\n    def __call__(self, y: np.ndarray):\n        n_trns = len(self.transforms)\n        trns_idx = np.random.choice(n_trns)\n        trns = self.transforms[trns_idx]\n        return trns(y)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-17T16:30:54.814681Z","iopub.execute_input":"2022-02-17T16:30:54.814888Z","iopub.status.idle":"2022-02-17T16:30:54.823655Z","shell.execute_reply.started":"2022-02-17T16:30:54.814863Z","shell.execute_reply":"2022-02-17T16:30:54.822822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1. AddGaussianNoise\n","metadata":{}},{"cell_type":"code","source":"class AddGaussianNoise(AudioTransform):\n    def __init__(self, always_apply=False, p=0.5, max_noise_amplitude=0.5, **kwargs):\n        super().__init__(always_apply, p)\n        self.noise_amplitude = (0.0, max_noise_amplitude)\n\n    def apply(self, y: np.ndarray, **params):\n        # 一様分布からサンプルを抽出\n        noise_amplitude = np.random.uniform(*self.noise_amplitude)\n        # 標準正規分布から出力値分をランダムで出力\n        noise = np.random.randn(len(y))\n        # 拡張\n        augmented = (y + noise * noise_amplitude).astype(y.dtype)\n        return augmented","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:54.825000Z","iopub.execute_input":"2022-02-17T16:30:54.825253Z","iopub.status.idle":"2022-02-17T16:30:54.836537Z","shell.execute_reply.started":"2022-02-17T16:30:54.825224Z","shell.execute_reply":"2022-02-17T16:30:54.835780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = AddGaussianNoise(always_apply=True, max_noise_amplitude=0.05)\ny_gaussian_added = transform(y)\n\n# 拡張結果を出力\nAudio(y_gaussian_added, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:54.837854Z","iopub.execute_input":"2022-02-17T16:30:54.838072Z","iopub.status.idle":"2022-02-17T16:30:54.869402Z","shell.execute_reply.started":"2022-02-17T16:30:54.838045Z","shell.execute_reply":"2022-02-17T16:30:54.868731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 音声波形の表示\nx = range(len(y_gaussian_added))\nplt.plot(x, y_gaussian_added)\nplt.plot(x, y_gaussian_added, color='blue')\nplt.legend(loc='upper center')\nplt.grid()\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:54.870549Z","iopub.execute_input":"2022-02-17T16:30:54.870907Z","iopub.status.idle":"2022-02-17T16:30:55.370037Z","shell.execute_reply.started":"2022-02-17T16:30:54.870867Z","shell.execute_reply":"2022-02-17T16:30:55.369444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# メル尺度のスペクトログラムの算出\nmel_spec = librosa.feature.melspectrogram(y=y_gaussian_added, sr=sr, n_mels=128)\n# パワースペクトログラム（振幅の2乗）をデシベル（dB）単位に変換\n# ログスケールに変換\nmelspec = librosa.power_to_db(mel_spec)\n# スペクトログラム/クロマトグラム/ cqt /などを表示\nlibrosa.display.specshow(melspec, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:55.371206Z","iopub.execute_input":"2022-02-17T16:30:55.371838Z","iopub.status.idle":"2022-02-17T16:30:55.786090Z","shell.execute_reply.started":"2022-02-17T16:30:55.371807Z","shell.execute_reply":"2022-02-17T16:30:55.785172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2. GaussianNoiseSNR","metadata":{}},{"cell_type":"code","source":"class GaussianNoiseSNR(AudioTransform):\n    def __init__(self, always_apply=False, p=0.5, min_snr=5.0, max_snr=20.0, **kwargs):\n        super().__init__(always_apply, p)\n        # 5\n        self.min_snr = min_snr\n        # 20\n        self.max_snr = max_snr\n\n    def apply(self, y: np.ndarray, **params):\n        snr = np.random.uniform(self.min_snr, self.max_snr)\n        a_signal = np.sqrt(y ** 2).max()\n        a_noise = a_signal / (10 ** (snr / 20))\n\n        white_noise = np.random.randn(len(y))\n        a_white = np.sqrt(white_noise ** 2).max()\n        augmented = (y + white_noise * 1 / a_white * a_noise).astype(y.dtype)\n        return augmented","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:55.787492Z","iopub.execute_input":"2022-02-17T16:30:55.788164Z","iopub.status.idle":"2022-02-17T16:30:55.797664Z","shell.execute_reply.started":"2022-02-17T16:30:55.788096Z","shell.execute_reply":"2022-02-17T16:30:55.796853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = GaussianNoiseSNR(always_apply=True, min_snr=5, max_snr=20)\ny_gaussian_snr = transform(y)\nAudio(y_gaussian_snr, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:55.798576Z","iopub.execute_input":"2022-02-17T16:30:55.798776Z","iopub.status.idle":"2022-02-17T16:30:55.831688Z","shell.execute_reply.started":"2022-02-17T16:30:55.798751Z","shell.execute_reply":"2022-02-17T16:30:55.830828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = range(len(y_gaussian_snr))\nplt.plot(x, y_gaussian_snr)\nplt.plot(x, y_gaussian_snr, color='blue')\nplt.legend(loc='upper center')\nplt.grid()\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:55.833031Z","iopub.execute_input":"2022-02-17T16:30:55.833792Z","iopub.status.idle":"2022-02-17T16:30:56.258502Z","shell.execute_reply.started":"2022-02-17T16:30:55.833751Z","shell.execute_reply":"2022-02-17T16:30:56.257644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mel_spec = librosa.feature.melspectrogram(y=y_gaussian_snr, sr=sr, n_mels=128)\nmelspec = librosa.power_to_db(mel_spec)\nlibrosa.display.specshow(melspec, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:56.259669Z","iopub.execute_input":"2022-02-17T16:30:56.259897Z","iopub.status.idle":"2022-02-17T16:30:56.671180Z","shell.execute_reply.started":"2022-02-17T16:30:56.259868Z","shell.execute_reply":"2022-02-17T16:30:56.670312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. PinkNoiseSNR","metadata":{}},{"cell_type":"code","source":"!pip install colorednoise","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:30:56.672622Z","iopub.execute_input":"2022-02-17T16:30:56.672986Z","iopub.status.idle":"2022-02-17T16:31:03.781304Z","shell.execute_reply.started":"2022-02-17T16:30:56.672946Z","shell.execute_reply":"2022-02-17T16:31:03.780254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import colorednoise as cn\n\nclass PinkNoiseSNR(AudioTransform):\n    def __init__(self, always_apply=False, p=0.5, min_snr=5.0, max_snr=20.0, **kwargs):\n        super().__init__(always_apply, p)\n\n        self.min_snr = min_snr\n        self.max_snr = max_snr\n\n    def apply(self, y: np.ndarray, **params):\n        snr = np.random.uniform(self.min_snr, self.max_snr)\n        a_signal = np.sqrt(y ** 2).max()\n        a_noise = a_signal / (10 ** (snr / 20))\n\n        pink_noise = cn.powerlaw_psd_gaussian(1, len(y))\n        a_pink = np.sqrt(pink_noise ** 2).max()\n        augmented = (y + pink_noise * 1 / a_pink * a_noise).astype(y.dtype)\n        return augmented","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:03.783473Z","iopub.execute_input":"2022-02-17T16:31:03.783764Z","iopub.status.idle":"2022-02-17T16:31:03.793424Z","shell.execute_reply.started":"2022-02-17T16:31:03.783730Z","shell.execute_reply":"2022-02-17T16:31:03.792134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = PinkNoiseSNR(always_apply=True, min_snr=5.0, max_snr=20.0)\ny_pink_noise = transform(y)\nAudio(y_pink_noise, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:03.794662Z","iopub.execute_input":"2022-02-17T16:31:03.795540Z","iopub.status.idle":"2022-02-17T16:31:03.854447Z","shell.execute_reply.started":"2022-02-17T16:31:03.795486Z","shell.execute_reply":"2022-02-17T16:31:03.853565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = range(len(y_pink_noise))\nplt.plot(x, y_pink_noise)\nplt.plot(x, y_pink_noise, color='blue')\nplt.legend(loc='upper center')\nplt.grid()\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:03.855718Z","iopub.execute_input":"2022-02-17T16:31:03.856026Z","iopub.status.idle":"2022-02-17T16:31:04.333449Z","shell.execute_reply.started":"2022-02-17T16:31:03.855994Z","shell.execute_reply":"2022-02-17T16:31:04.332607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mel_spec = librosa.feature.melspectrogram(y=y_pink_noise, sr=sr, n_mels=128)\nmelspec = librosa.power_to_db(mel_spec)\nlibrosa.display.specshow(melspec, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:04.334702Z","iopub.execute_input":"2022-02-17T16:31:04.334943Z","iopub.status.idle":"2022-02-17T16:31:04.743607Z","shell.execute_reply.started":"2022-02-17T16:31:04.334915Z","shell.execute_reply":"2022-02-17T16:31:04.743039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4. PitchShift","metadata":{}},{"cell_type":"code","source":"class PitchShift(AudioTransform):\n    def __init__(self, always_apply=False, p=0.5, max_steps=5, sr=32000):\n        super().__init__(always_apply, p)\n\n        self.max_steps = max_steps\n        self.sr = sr\n\n    def apply(self, y: np.ndarray, **params):\n        n_steps = np.random.randint(-self.max_steps, self.max_steps)\n        augmented = librosa.effects.pitch_shift(y, sr=self.sr, n_steps=n_steps)\n        return augmented","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:04.744770Z","iopub.execute_input":"2022-02-17T16:31:04.745147Z","iopub.status.idle":"2022-02-17T16:31:04.750132Z","shell.execute_reply.started":"2022-02-17T16:31:04.745091Z","shell.execute_reply":"2022-02-17T16:31:04.749619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = PitchShift(always_apply=True, max_steps=2, sr=sr)\ny_pitch_shift = transform(y)\nAudio(y_pitch_shift, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:04.751153Z","iopub.execute_input":"2022-02-17T16:31:04.751488Z","iopub.status.idle":"2022-02-17T16:31:05.082577Z","shell.execute_reply.started":"2022-02-17T16:31:04.751446Z","shell.execute_reply":"2022-02-17T16:31:05.081725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = range(len(y_pitch_shift))\nplt.plot(x, y_pitch_shift)\nplt.plot(x, y_pitch_shift, color='blue')\nplt.legend(loc='upper center')\nplt.grid()\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:05.087148Z","iopub.execute_input":"2022-02-17T16:31:05.087399Z","iopub.status.idle":"2022-02-17T16:31:05.572019Z","shell.execute_reply.started":"2022-02-17T16:31:05.087369Z","shell.execute_reply":"2022-02-17T16:31:05.571436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mel_spec = librosa.feature.melspectrogram(y=y_pitch_shift, sr=sr, n_mels=128)\nmelspec = librosa.power_to_db(mel_spec)\nlibrosa.display.specshow(melspec, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:05.573065Z","iopub.execute_input":"2022-02-17T16:31:05.573421Z","iopub.status.idle":"2022-02-17T16:31:05.994157Z","shell.execute_reply.started":"2022-02-17T16:31:05.573392Z","shell.execute_reply":"2022-02-17T16:31:05.993556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 5. TimeStretch\n","metadata":{}},{"cell_type":"code","source":"class TimeStretch(AudioTransform):\n    def __init__(self, always_apply=False, p=0.5, max_rate=1.2):\n        super().__init__(always_apply, p)\n\n        self.max_rate = max_rate\n\n    def apply(self, y: np.ndarray, **params):\n        rate = np.random.uniform(0, self.max_rate)\n        augmented = librosa.effects.time_stretch(y, rate=rate)\n        return augmented","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:05.995390Z","iopub.execute_input":"2022-02-17T16:31:05.995787Z","iopub.status.idle":"2022-02-17T16:31:06.001006Z","shell.execute_reply.started":"2022-02-17T16:31:05.995756Z","shell.execute_reply":"2022-02-17T16:31:06.000367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = TimeStretch(always_apply=True, max_rate=2.0)\ny_time_stretch = transform(y)\nAudio(y_time_stretch, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:06.002073Z","iopub.execute_input":"2022-02-17T16:31:06.002302Z","iopub.status.idle":"2022-02-17T16:31:06.172492Z","shell.execute_reply.started":"2022-02-17T16:31:06.002277Z","shell.execute_reply":"2022-02-17T16:31:06.171627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mel_spec = librosa.feature.melspectrogram(y=y_time_stretch, sr=sr, n_mels=128)\nmelspec = librosa.power_to_db(mel_spec)\nlibrosa.display.specshow(melspec, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:06.173565Z","iopub.execute_input":"2022-02-17T16:31:06.174328Z","iopub.status.idle":"2022-02-17T16:31:06.643007Z","shell.execute_reply.started":"2022-02-17T16:31:06.174294Z","shell.execute_reply":"2022-02-17T16:31:06.642168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 6. TimeShift","metadata":{}},{"cell_type":"code","source":"class TimeShift(AudioTransform):\n    def __init__(self, always_apply=False, p=0.5, max_shift_second=2, sr=32000, padding_mode=\"replace\"):\n        super().__init__(always_apply, p)\n    \n        assert padding_mode in [\"replace\", \"zero\"], \"`padding_mode` must be either 'replace' or 'zero'\"\n        self.max_shift_second = max_shift_second\n        self.sr = sr\n        self.padding_mode = padding_mode\n\n    def apply(self, y: np.ndarray, **params):\n        shift = np.random.randint(-self.sr * self.max_shift_second, self.sr * self.max_shift_second)\n        augmented = np.roll(y, shift)\n        if self.padding_mode == \"zero\":\n            if shift > 0:\n                augmented[:shift] = 0\n            else:\n                augmented[shift:] = 0\n        return augmented","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:06.644749Z","iopub.execute_input":"2022-02-17T16:31:06.645055Z","iopub.status.idle":"2022-02-17T16:31:06.656909Z","shell.execute_reply.started":"2022-02-17T16:31:06.645015Z","shell.execute_reply":"2022-02-17T16:31:06.653186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = TimeShift(always_apply=True, max_shift_second=4, sr=sr)\ny_time_shifted = transform(y)\nAudio(y_time_shifted, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:06.658550Z","iopub.execute_input":"2022-02-17T16:31:06.659039Z","iopub.status.idle":"2022-02-17T16:31:06.679147Z","shell.execute_reply.started":"2022-02-17T16:31:06.658990Z","shell.execute_reply":"2022-02-17T16:31:06.678554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = range(len(y_time_shifted))\nplt.plot(x, y_time_shifted)\nplt.plot(x, y_time_shifted, color='blue')\nplt.legend(loc='upper center')\nplt.grid()\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:06.680047Z","iopub.execute_input":"2022-02-17T16:31:06.680605Z","iopub.status.idle":"2022-02-17T16:31:07.065295Z","shell.execute_reply.started":"2022-02-17T16:31:06.680571Z","shell.execute_reply":"2022-02-17T16:31:07.064696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mel_spec = librosa.feature.melspectrogram(y=y_time_shifted, sr=sr, n_mels=128)\nmelspec = librosa.power_to_db(mel_spec)\nlibrosa.display.specshow(melspec, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:07.066212Z","iopub.execute_input":"2022-02-17T16:31:07.066567Z","iopub.status.idle":"2022-02-17T16:31:07.486536Z","shell.execute_reply.started":"2022-02-17T16:31:07.066532Z","shell.execute_reply":"2022-02-17T16:31:07.485704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 7. VolumeControl","metadata":{}},{"cell_type":"code","source":"class VolumeControl(AudioTransform):\n    def __init__(self, always_apply=False, p=0.5, db_limit=10, mode='uniform'):\n        super().__init__(always_apply, p)\n\n        assert mode in [\"uniform\", \"fade\", \"fade\", \"cosine\", \"sine\"], \\\n            \"`mode` must be one of 'uniform', 'fade', 'cosine', 'sine'\"\n        \n        self.db_limit= db_limit\n        self.mode = mode\n\n    def apply(self, y: np.ndarray, **params):\n        db = np.random.uniform(-self.db_limit, self.db_limit)\n        if self.mode == 'uniform':\n            db_translated = 10 ** (db / 20)\n        elif self.mode == 'fade':\n            lin = np.arange(len(y))[::-1] / (len(y) - 1)\n            db_translated = 10 ** (db * lin / 20)\n        elif self.mode == 'cosine':\n            cosine = np.cos(np.arange(len(y)) / len(y) * np.pi * 2)\n            db_translated = 10 ** (db * cosine / 20)\n        else:\n            sine = np.sin(np.arange(len(y)) / len(y) * np.pi * 2)\n            db_translated = 10 ** (db * sine / 20)\n        augmented = y * db_translated\n        return augmented","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:07.488022Z","iopub.execute_input":"2022-02-17T16:31:07.488720Z","iopub.status.idle":"2022-02-17T16:31:07.500232Z","shell.execute_reply.started":"2022-02-17T16:31:07.488675Z","shell.execute_reply":"2022-02-17T16:31:07.499616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transform = VolumeControl(always_apply=True, mode='sine')\ny_volume_controlled = transform(y)\nAudio(y_volume_controlled, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:07.501710Z","iopub.execute_input":"2022-02-17T16:31:07.502235Z","iopub.status.idle":"2022-02-17T16:31:07.535679Z","shell.execute_reply.started":"2022-02-17T16:31:07.502199Z","shell.execute_reply":"2022-02-17T16:31:07.534794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = range(len(y_volume_controlled))\nplt.plot(x, y_volume_controlled)\nplt.plot(x, y_volume_controlled, color='blue')\nplt.legend(loc='upper center')\nplt.grid()\nplt.grid()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:07.537211Z","iopub.execute_input":"2022-02-17T16:31:07.537571Z","iopub.status.idle":"2022-02-17T16:31:07.931417Z","shell.execute_reply.started":"2022-02-17T16:31:07.537533Z","shell.execute_reply":"2022-02-17T16:31:07.929737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mel_spec = librosa.feature.melspectrogram(y=y_volume_controlled, sr=sr, n_mels=128)\nmelspec = librosa.power_to_db(mel_spec)\nlibrosa.display.specshow(melspec, sr=sr, x_axis='time', y_axis='mel')\nplt.colorbar()","metadata":{"execution":{"iopub.status.busy":"2022-02-17T16:31:07.932470Z","iopub.execute_input":"2022-02-17T16:31:07.932673Z","iopub.status.idle":"2022-02-17T16:31:08.369624Z","shell.execute_reply.started":"2022-02-17T16:31:07.932647Z","shell.execute_reply":"2022-02-17T16:31:08.369009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}