{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os \nimport numpy as np, pandas as pd \n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom glob import glob\n\nimport warnings \nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:31.954242Z","iopub.execute_input":"2024-05-30T10:13:31.954621Z","iopub.status.idle":"2024-05-30T10:13:31.960760Z","shell.execute_reply.started":"2024-05-30T10:13:31.954593Z","shell.execute_reply":"2024-05-30T10:13:31.959473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/birdclef-2024/train_metadata.csv')\nprint(f'Shape of DataFrame: {df.shape}')\nprint(display(df))","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:45:39.880564Z","iopub.execute_input":"2024-05-30T10:45:39.881038Z","iopub.status.idle":"2024-05-30T10:45:40.020044Z","shell.execute_reply.started":"2024-05-30T10:45:39.880992Z","shell.execute_reply":"2024-05-30T10:45:40.018986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'There are {df[\"primary_label\"].nunique()} Primary Label in DataFrame')\nprint(\"I'm gonna make spectrogram from asbfly's audio\")","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:32.128702Z","iopub.execute_input":"2024-05-30T10:13:32.129166Z","iopub.status.idle":"2024-05-30T10:13:32.138760Z","shell.execute_reply.started":"2024-05-30T10:13:32.129128Z","shell.execute_reply":"2024-05-30T10:13:32.137593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import IPython.display as ipd\n\nasbfly_path = glob('/kaggle/input/birdclef-2024/train_audio/asbfly/*')\nprint(\"This Audio is asbfly's Song\")\nipd.Audio(asbfly_path[0])","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:32.139973Z","iopub.execute_input":"2024-05-30T10:13:32.140360Z","iopub.status.idle":"2024-05-30T10:13:32.168537Z","shell.execute_reply.started":"2024-05-30T10:13:32.140324Z","shell.execute_reply":"2024-05-30T10:13:32.167465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Load with Librosa**","metadata":{}},{"cell_type":"code","source":"import librosa \n\ny, sr = librosa.load(asbfly_path[0], sr=None)\n\nplt.figure(figsize=(12,6))\nplt.title(\"Asbfly's audio loaded by librosa\")\npd.Series(y).plot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:32.171704Z","iopub.execute_input":"2024-05-30T10:13:32.172162Z","iopub.status.idle":"2024-05-30T10:13:33.116982Z","shell.execute_reply.started":"2024-05-30T10:13:32.172105Z","shell.execute_reply":"2024-05-30T10:13:33.115773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Load with Torch Audio**","metadata":{}},{"cell_type":"code","source":"import torchaudio\n\ny, sr = torchaudio.load(asbfly_path[0]) # Channel Size X  Audio\n\nplt.figure(figsize=(12,6))\nplt.title(\"Asbfly's audio loaded by torchaudio\")\npd.Series(y.squeeze().numpy()).plot()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:33.118564Z","iopub.execute_input":"2024-05-30T10:13:33.119049Z","iopub.status.idle":"2024-05-30T10:13:34.100420Z","shell.execute_reply.started":"2024-05-30T10:13:33.119008Z","shell.execute_reply":"2024-05-30T10:13:34.099158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"directory_path = '/bird_spectrograms/'\nif not os.path.exists(directory_path):\n    os.makedirs(directory_path)","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:56:50.685263Z","iopub.execute_input":"2024-05-30T10:56:50.685684Z","iopub.status.idle":"2024-05-30T10:56:50.691770Z","shell.execute_reply.started":"2024-05-30T10:56:50.685652Z","shell.execute_reply":"2024-05-30T10:56:50.690355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def spec_from_audio(path):\n    \n    audio, sr = librosa.load(path, sr=None)\n    \n    m = np.nanmean(audio)\n    audio = np.nan_to_num(audio, nan=m)\n    \n    raw_spec = librosa.feature.melspectrogram(y=audio, sr=sr, hop_length = len(audio)//256, n_mels = 256)\n    \n    width = (raw_spec.shape[1]//32)*32\n    mel_spec_db = librosa.power_to_db(raw_spec, ref=np.max)[:,:width]\n\n    mel_spec_db = (mel_spec_db+40)/40 # -80~0db -> ~40~40db\n    return mel_spec_db","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:34.110484Z","iopub.execute_input":"2024-05-30T10:13:34.110852Z","iopub.status.idle":"2024-05-30T10:13:34.122441Z","shell.execute_reply.started":"2024-05-30T10:13:34.110822Z","shell.execute_reply":"2024-05-30T10:13:34.121362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create Spectrograms with Librosa\n___\n![](https://encrypted-tbn0.gstatic.com/images?q=tbn:ANd9GcRCsNBr8aegj2WZcXA9XoduSCbuhIXCSuPptA&s)","metadata":{}},{"cell_type":"markdown","source":"### Display Spectrogram","metadata":{}},{"cell_type":"code","source":"audio, sr = librosa.load(asbfly_path[0], sr=None)\n# Fill Nan \nm = np.nanmean(audio)\naudio = np.nan_to_num(audio, nan=m)\n    \n# Audio is already -1~1 \n# There's no standardization & removing Outlier\n\nplt.figure(figsize=(12,6))\n# Raw Spectrogram\nplt.subplot(1,3,1)\nraw_spec = librosa.feature.melspectrogram(y=audio, sr=sr)\nplt.title('Raw Spectrogram')\nlibrosa.display.specshow(raw_spec, sr=sr, x_axis='time', y_axis='mel')\n\n# Log Spectrogram\nplt.subplot(1,3,2)\nmel_spec_db = librosa.power_to_db(raw_spec, ref=np.max)\nplt.title('Before Standardization')\nlibrosa.display.specshow(mel_spec_db, sr=sr, x_axis='time', y_axis='mel')\n\n# Standardize to -1 to 1\nplt.subplot(1,3,3)\nmel_spec_db = (mel_spec_db+40)/40 # -80~0db -> ~40~40db\nplt.title('After Standardization')\nlibrosa.display.specshow(mel_spec_db, sr=sr, x_axis='time', y_axis='mel')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:34.123922Z","iopub.execute_input":"2024-05-30T10:13:34.124330Z","iopub.status.idle":"2024-05-30T10:13:37.076541Z","shell.execute_reply.started":"2024-05-30T10:13:34.124300Z","shell.execute_reply":"2024-05-30T10:13:37.075307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's Make Image(SIZE: 256x256)","metadata":{}},{"cell_type":"markdown","source":"I'm gonna use library librosa to create spectrograms.\n\n`Spectrograms of size = 256X256(freqXtime)`\n\nThe main function is \n\n    mel_spec = librsoa.feature.melspectrogram(y=x, sr=200, hop_length=len(x)//256, n_fft=1024, n_mels=128, fmin=0, fmax=20, win_length=128\n    \n* `y` is the input time series signal \n* `sr` is the sampling frequency.\n* `hop_length` produces image with `width=len(x)/hop_length`\n* `n_fft` controls vertical resolution and quality of spectrogram\n* `n_mels` produces image with `height=n_mels`\n* `f_min` is smallest frequency in our spectrogram\n* `f_max` is largest frequency in our spectrogram\n* `win_length` controls hortizonal resolution and quality of spectrogram","metadata":{}},{"cell_type":"code","source":"audio, sr = librosa.load(asbfly_path[0], sr=None)\n# Fill Nan \nm = np.nanmean(audio)\naudio = np.nan_to_num(audio, nan=m)\n    \n# Audio is already -1~1 \n# There's no standardization & removing Outlier\n\nplt.figure(figsize=(12,12))\n# Raw Spectrogram\nplt.subplot(1,3,1)\nraw_spec = librosa.feature.melspectrogram(y=audio, sr=sr, hop_length=len(audio)//256, n_mels=256)\nwidth = (raw_spec.shape[1]//32)*32 # 257 -> 256\nplt.title('Raw Spectrogram')\nplt.imshow(raw_spec, origin='lower')\n\n# Log Spectrogram\nplt.subplot(1,3,2)\nmel_spec_db = librosa.power_to_db(raw_spec, ref=np.max)[:,:width]\nplt.title('Before Standardization')\nplt.imshow(mel_spec_db, origin='lower')\n\n# Standardize to -1 to 1\nplt.subplot(1,3,3)\nmel_spec_db = (mel_spec_db+40)/40 # -80~0db -> ~40~40db\nplt.title('After Standardization')\nplt.imshow(mel_spec_db, origin='lower')\nplt.tight_layout()","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:43.434439Z","iopub.execute_input":"2024-05-30T10:13:43.434833Z","iopub.status.idle":"2024-05-30T10:13:44.539347Z","shell.execute_reply.started":"2024-05-30T10:13:43.434803Z","shell.execute_reply":"2024-05-30T10:13:44.538168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Spectrogram Augmentation","metadata":{}},{"cell_type":"markdown","source":"#### 1. XYMasking","metadata":{}},{"cell_type":"code","source":"import albumentations as albu\n\nparams = {\n    \"num_masks_x\": 2,\n    'num_masks_y': 1,\n    'mask_x_length': (10,20),\n    'mask_y_length': (10,20),\n    'fill_value': 1,   \n}\n\ncomposition = albu.Compose([albu.XYMasking(**params, p=1.0)])\naugmented_spec_1 = composition(image=mel_spec_db)['image']\naugmented_spec_2 = composition(image=mel_spec_db)['image']\n\nplt.figure(figsize=(12,6))\nplt.subplot(1,2,1)\nplt.imshow(augmented_spec_1, origin='lower')\nplt.subplot(1,2,2)\nplt.imshow(augmented_spec_2, origin='lower')","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:48.198491Z","iopub.execute_input":"2024-05-30T10:13:48.198918Z","iopub.status.idle":"2024-05-30T10:13:48.884510Z","shell.execute_reply.started":"2024-05-30T10:13:48.198885Z","shell.execute_reply":"2024-05-30T10:13:48.883250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 2. MixUp","metadata":{}},{"cell_type":"markdown","source":"**Beta Distribution for alpha 0.5, 1.0, 2.0**","metadata":{}},{"cell_type":"code","source":"i = 1\nplt.figure(figsize=(12,6))\n\nfor alpha in [0.5, 1.0, 2.0]:\n    x = [np.random.beta(alpha, alpha) for i in range(2000)]\n    plt.subplot(1,3,i)\n    plt.title(f\"Beta distribution for alpha: {alpha}\")\n    plt.hist(x)\n    i += 1\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:13:51.264042Z","iopub.execute_input":"2024-05-30T10:13:51.264480Z","iopub.status.idle":"2024-05-30T10:13:52.055099Z","shell.execute_reply.started":"2024-05-30T10:13:51.264447Z","shell.execute_reply":"2024-05-30T10:13:52.053967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path1 = asbfly_path[1]\npath2 = asbfly_path[2]\n\nplt.figure(figsize=(12,6))\nplt.subplot(1,3,1)\nspec1 = spec_from_audio(path1)\nplt.title(\"Spectrogram 1\")\nplt.imshow(spec1, origin='lower')\nplt.subplot(1,3,2)\nspec2 = spec_from_audio(path2)\nplt.title(\"Spectrogram 2\")\nplt.imshow(spec2, origin='lower')\n        \n# lam = np.random.beta(0.5,0.5)\n        \nmixed_image = 0.5 * spec1 + 0.5 * spec2 \n\nplt.subplot(1,3,3)\nplt.title('Mix up Spectrogram')\nplt.imshow(mixed_image, origin='lower')","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:20:35.218002Z","iopub.execute_input":"2024-05-30T10:20:35.218435Z","iopub.status.idle":"2024-05-30T10:20:36.300251Z","shell.execute_reply.started":"2024-05-30T10:20:35.218403Z","shell.execute_reply":"2024-05-30T10:20:36.299043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 3. CutMix","metadata":{}},{"cell_type":"code","source":"path1 = asbfly_path[3]\npath2 = asbfly_path[4]\n\nplt.figure(figsize=(12,6))\nplt.subplot(1,3,1)\nspec1 = spec_from_audio(path1)\nplt.title(\"Spectrogram 1\")\nplt.imshow(spec1, origin='lower')\nplt.subplot(1,3,2)\nspec2 = spec_from_audio(path2)\nplt.title(\"Spectrogram 2\")\nplt.imshow(spec2, origin='lower')\n        \nh, w = spec1.shape\nlam = np.random.beta(2.0,2.0)\ncut_w = int(w * lam)\ncut_h = int(h * lam)\n\ncx = np.random.randint(w)\ncy = np.random.randint(h)\n\nbby1 = np.clip(cy-cut_h//2, 0, h) ; bby2= np.clip(cy+cut_h//2, 0, h) \nbbx1 = np.clip(cx-cut_w//2, 0, w) ; bbx2= np.clip(cx+cut_w//2, 0, w)\n\nmixed_image = spec1.copy()\nmixed_image[bby1:bby2, bbx1:bbx2] = spec2[bby1:bby2, bbx1:bbx2]\n\nplt.subplot(1,3,3)\nplt.title('CutMix Spectrogram')\nplt.imshow(mixed_image, origin='lower')","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:35:35.224465Z","iopub.execute_input":"2024-05-30T10:35:35.225437Z","iopub.status.idle":"2024-05-30T10:35:36.470265Z","shell.execute_reply.started":"2024-05-30T10:35:35.225397Z","shell.execute_reply":"2024-05-30T10:35:36.469128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Save to Disk\nIn this Case, I just only saved `asbfly`","metadata":{}},{"cell_type":"code","source":"%%time \n\nPATH = '/kaggle/input/birdclef-2024/train_audio/'\nFILE_NAME = df[df['primary_label'] == 'asbfly'].filename.unique()\nall_specs = {}\n\nfor i, file_name in enumerate(FILE_NAME):\n    if (i%100==0)&(i!=0): print(i,', ', end='')\n    \n    spec = spec_from_audio(f'{PATH}{file_name}')\n    \n    # save to disk\n    file_name = file_name.split('/')[0]\n    np.save(f'{directory_path}{file_name}',spec)\n    all_specs[file_name] = spec\n                         \nnp.save('Bird_specs', all_specs)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-30T10:58:50.904497Z","iopub.execute_input":"2024-05-30T10:58:50.904912Z","iopub.status.idle":"2024-05-30T10:59:03.734080Z","shell.execute_reply.started":"2024-05-30T10:58:50.904881Z","shell.execute_reply":"2024-05-30T10:59:03.732747Z"},"trusted":true},"execution_count":null,"outputs":[]}]}