{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8090934,"sourceType":"datasetVersion","datasetId":4776799}],"dockerImageVersionId":30683,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Configuration","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport math\nimport librosa\nimport pandas as pd\nimport numpy as np\nfrom tqdm.notebook import tqdm\n\nimport torch","metadata":{"execution":{"iopub.status.busy":"2024-04-12T02:34:53.251618Z","iopub.execute_input":"2024-04-12T02:34:53.251992Z","iopub.status.idle":"2024-04-12T02:34:53.258071Z","shell.execute_reply.started":"2024-04-12T02:34:53.251963Z","shell.execute_reply":"2024-04-12T02:34:53.256874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    \n    # == global config ==\n    OUTPUT_DIR = '/kaggle/working/'  # output folder\n    \n    # == data config ==\n    DATA_ROOT = '/kaggle/input/birdclef-2024'  # root folder\n    FS = 32000  # sample rate\n    N_FFT = 1095  # n FFT of Spec.\n    WIN_SIZE = 412  # WIN_SIZE of Spec.\n    WIN_LAP = 100  # overlap of Spec.\n    MIN_FREQ = 40  # min frequency\n    MAX_FREQ = 15000  # max frequency\n    \n    N_MAX = 10  # max number of samples","metadata":{"execution":{"iopub.status.busy":"2024-04-12T02:34:03.733272Z","iopub.execute_input":"2024-04-12T02:34:03.733691Z","iopub.status.idle":"2024-04-12T02:34:03.738893Z","shell.execute_reply.started":"2024-04-12T02:34:03.733665Z","shell.execute_reply":"2024-04-12T02:34:03.737864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Metadata","metadata":{}},{"cell_type":"code","source":"# labels\nlabel_list = sorted(os.listdir(os.path.join(config.DATA_ROOT, 'train_audio')))\nlabel_id_list = list(range(len(label_list)))\nlabel2id = dict(zip(label_list, label_id_list))\nid2label = dict(zip(label_id_list, label_list))","metadata":{"execution":{"iopub.status.busy":"2024-04-12T02:34:03.740005Z","iopub.execute_input":"2024-04-12T02:34:03.740287Z","iopub.status.idle":"2024-04-12T02:34:03.767744Z","shell.execute_reply.started":"2024-04-12T02:34:03.740262Z","shell.execute_reply":"2024-04-12T02:34:03.767034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.environ[\"CUDA_VISIBLE_DEVICES\"] = \"0,1\"\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\nprint('Using', torch.cuda.device_count(), 'GPU(s)')","metadata":{"execution":{"iopub.status.busy":"2024-04-12T02:34:03.768685Z","iopub.execute_input":"2024-04-12T02:34:03.768950Z","iopub.status.idle":"2024-04-12T02:34:03.851955Z","shell.execute_reply.started":"2024-04-12T02:34:03.768927Z","shell.execute_reply":"2024-04-12T02:34:03.851046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_df = pd.read_csv(f'{config.DATA_ROOT}/train_metadata.csv')\nmetadata_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-12T02:34:03.854578Z","iopub.execute_input":"2024-04-12T02:34:03.855196Z","iopub.status.idle":"2024-04-12T02:34:04.112622Z","shell.execute_reply.started":"2024-04-12T02:34:03.855167Z","shell.execute_reply":"2024-04-12T02:34:04.111605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = metadata_df[['primary_label', 'rating', 'filename']].copy()\n\n# create target\ntrain_df['target'] = train_df.primary_label.map(label2id)\n# create filepath\ntrain_df['filepath'] = config.DATA_ROOT + '/train_audio/' + train_df.filename\n# create new sample name\ntrain_df['samplename'] = train_df.filename.map(lambda x: x.split('/')[0] + '-' + x.split('/')[-1].split('.')[0])\n\nprint(f'find {len(train_df)} samples')\n\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-12T02:34:04.113731Z","iopub.execute_input":"2024-04-12T02:34:04.114036Z","iopub.status.idle":"2024-04-12T02:34:04.179461Z","shell.execute_reply.started":"2024-04-12T02:34:04.114011Z","shell.execute_reply":"2024-04-12T02:34:04.178599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Pre-processing","metadata":{}},{"cell_type":"code","source":"def oog2spec_via_cupy(audio_data):\n    \n    import cupy as cp\n    from cupyx.scipy import signal as cupy_signal\n    \n    audio_data = cp.array(audio_data)\n    \n    # handles NaNs\n    mean_signal = cp.nanmean(audio_data)\n    audio_data = cp.nan_to_num(audio_data, nan=mean_signal) if cp.isnan(audio_data).mean() < 1 else cp.zeros_like(audio_data)\n    \n    # to spec.\n    frequencies, times, spec_data = cupy_signal.spectrogram(\n        audio_data, \n        fs=config.FS, \n        nfft=config.N_FFT, \n        nperseg=config.WIN_SIZE, \n        noverlap=config.WIN_LAP, \n        window='hann'\n    )\n    \n    # Filter frequency range\n    valid_freq = (frequencies >= config.MIN_FREQ) & (frequencies <= config.MAX_FREQ)\n    spec_data = spec_data[valid_freq, :]\n    \n    # Log\n    spec_data = cp.log10(spec_data + 1e-20)\n    \n    # min/max normalize\n    spec_data = spec_data - spec_data.min()\n    spec_data = spec_data / spec_data.max()\n    \n    return spec_data.get()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-12T02:34:04.180663Z","iopub.execute_input":"2024-04-12T02:34:04.180955Z","iopub.status.idle":"2024-04-12T02:34:04.188452Z","shell.execute_reply.started":"2024-04-12T02:34:04.180930Z","shell.execute_reply":"2024-04-12T02:34:04.187605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_bird_data = dict()\nfor i, row_metadata in tqdm(train_df.iterrows()):\n\n    # load ogg\n    audio_data, _ = librosa.load(row_metadata.filepath, sr=config.FS)\n\n    # crop\n    n_copy = math.ceil(5 * config.FS / len(audio_data))\n    if n_copy > 1: audio_data = np.concatenate([audio_data]*n_copy)\n\n    start_idx = int(len(audio_data) / 2 - 2.5 * config.FS)\n    end_idx = int(start_idx + 5.0 * config.FS)\n    input_audio = audio_data[start_idx:end_idx]\n\n    # ogg to spec.\n    input_spec = oog2spec_via_cupy(input_audio)\n\n    input_spec = cv2.resize(input_spec, (256, 256), interpolation=cv2.INTER_AREA)\n\n    all_bird_data[row_metadata.samplename] = input_spec.astype(np.float32)\n    \n    if i == config.N_MAX:\n        break\n\n# save to file\n# np.save(os.path.join(config.OUTPUT_DIR, f'spec_center_5sec_256_256.npy'), all_bird_data)","metadata":{"execution":{"iopub.status.busy":"2024-04-12T02:34:57.673646Z","iopub.execute_input":"2024-04-12T02:34:57.674331Z","iopub.status.idle":"2024-04-12T02:36:49.253755Z","shell.execute_reply.started":"2024-04-12T02:34:57.674282Z","shell.execute_reply":"2024-04-12T02:36:49.252758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Plot","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n\nplot_keys = list(all_bird_data.keys())[:4]\n\nfig = plt.figure(figsize=(12,12))\nfor i in range(4):\n    img = all_bird_data[plot_keys[i]]\n    \n    ax = fig.add_subplot(2, 2, i + 1, xticks=[], yticks=[])\n    ax.imshow(img, cmap='jet')\n    ax.set_title(f'ID: {plot_keys[i]}')\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-12T02:38:43.313624Z","iopub.execute_input":"2024-04-12T02:38:43.314257Z","iopub.status.idle":"2024-04-12T02:38:44.531124Z","shell.execute_reply.started":"2024-04-12T02:38:43.314225Z","shell.execute_reply":"2024-04-12T02:38:44.530171Z"},"trusted":true},"execution_count":null,"outputs":[]}]}