{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from fastai.vision.all import *\nimport timm\nfrom sklearn.model_selection import train_test_split\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\nimport soundfile as sf\nimport librosa as lb","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-15T22:14:37.851436Z","iopub.execute_input":"2023-05-15T22:14:37.851911Z","iopub.status.idle":"2023-05-15T22:14:37.920598Z","shell.execute_reply.started":"2023-05-15T22:14:37.851868Z","shell.execute_reply":"2023-05-15T22:14:37.919229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = Path('/kaggle/input')\ntest_audio_path = path/'birdclef-2023/test_soundscapes'\ntest_img_path = Path('/kaggle/working/test_images')\ntest_audio_path.ls().sorted()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T22:14:37.923664Z","iopub.execute_input":"2023-05-15T22:14:37.924083Z","iopub.status.idle":"2023-05-15T22:14:37.937497Z","shell.execute_reply.started":"2023-05-15T22:14:37.924031Z","shell.execute_reply":"2023-05-15T22:14:37.936541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.DataFrame(\n     sorted([(path.stem, *path.stem.split(\"_\"), path) for path in Path(test_audio_path).glob(\"*.ogg\")]),\n    columns = [\"filename\", \"name\" ,\"id\", \"path\"]\n)\nprint(df_test.shape)\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T22:14:37.938875Z","iopub.execute_input":"2023-05-15T22:14:37.939219Z","iopub.status.idle":"2023-05-15T22:14:37.977014Z","shell.execute_reply.started":"2023-05-15T22:14:37.939185Z","shell.execute_reply":"2023-05-15T22:14:37.975598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# params\nsr = 32000\nn_mels = 128\nfmin = 0\nfmax = None or sr // 2\nduration = 5\naudio_length = duration * sr\nstep = None or audio_length\nres_type=\"kaiser_fast\"\nresample=True\n\n\ndef compute_melspec(y, sr, n_mels, fmin, fmax):\n    melspec = lb.feature.melspectrogram(y=y, sr=sr, n_mels=n_mels, fmin=fmin, fmax=fmax)\n    melspec = lb.power_to_db(melspec).astype(np.float32)\n    return melspec\n\n\ndef mono_to_color(X, eps=1e-6, mean=None, std=None):\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n    \n    _min, _max = X.min(), X.max()\n\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n\n    return V\n\ndef crop_or_pad(y, length, is_train=True, start=None):\n    if len(y) < length:\n        y = np.concatenate([y, np.zeros(length - len(y))])\n        \n        n_repeats = length // len(y)\n        epsilon = length % len(y)\n        \n        y = np.concatenate([y]*n_repeats + [y[:epsilon]])\n        \n    elif len(y) > length:\n        if not is_train:\n            start = start or 0\n        else:\n            start = start or np.random.randint(len(y) - length)\n\n        y = y[start:start + length]\n\n    return y\n\n\ndef audio_to_image(audio, sr, n_mels, fmin, fmax):\n    melspec = compute_melspec(audio, sr, n_mels, fmin, fmax) \n    image = mono_to_color(melspec)\n    return image\n    \n    \ndef create_test_imgs(aud, test_img_path, test_soundscape_path, audio_length, sr, resample, res_type, n_mels, fmin, fmax, step):\n    audio, orig_sr = sf.read(aud, dtype=\"float32\")\n\n    if resample and orig_sr != sr:\n        audio = lb.resample(audio, orig_sr, sr, res_type=res_type)\n\n    n_slices = (len(audio) + audio_length - 1) // audio_length\n    audios = [audio[i * audio_length: (i + 1) * audio_length] for i in range(n_slices - 1)]\n    audios.append(crop_or_pad(audio[(n_slices - 1) * audio_length:], length=audio_length))\n\n    for i, audio in enumerate(audios):\n        img = audio_to_image(audio=audio, sr=sr, n_mels=n_mels, fmin=fmin, fmax=fmax)\n        img = Image.fromarray(img)\n        img.save(test_img_path/f'{aud.stem}_{5 * (i + 1)}.png')\n\n            \ndef parallel_test_process(test_audio_path, test_img_path, audio_length, sr, resample, res_type, n_mels, fmin, fmax, step, n_jobs=-1):\n    soundscapes = test_audio_path.ls().sorted()\n    Parallel(n_jobs=n_jobs)(delayed(create_test_imgs)(soundscape, test_img_path, test_audio_path, audio_length, sr, resample, res_type, n_mels, fmin, fmax, step) for soundscape in tqdm(soundscapes))","metadata":{"execution":{"iopub.status.busy":"2023-05-15T22:14:37.979915Z","iopub.execute_input":"2023-05-15T22:14:37.980262Z","iopub.status.idle":"2023-05-15T22:14:38.006097Z","shell.execute_reply.started":"2023-05-15T22:14:37.980229Z","shell.execute_reply":"2023-05-15T22:14:38.004760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mkdir(test_img_path, exist_ok=True)\nparallel_test_process(test_audio_path, test_img_path, audio_length, sr, resample, res_type, n_mels, fmin, fmax, step)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T22:14:38.008130Z","iopub.execute_input":"2023-05-15T22:14:38.008476Z","iopub.status.idle":"2023-05-15T22:14:56.229616Z","shell.execute_reply.started":"2023-05-15T22:14:38.008444Z","shell.execute_reply":"2023-05-15T22:14:56.227959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def filter_images(file_list, file_name):\n    return [file for file in file_list if file.startswith(file_name)]\n\ndef sort_images(images):\n    return sorted(images, key=lambda x: int(x.split('_')[-1].split('.')[0]))\n\nfiles = os.listdir(test_img_path)\n\nfile_names = sorted([file_name.split('.')[0] for file_name in os.listdir(test_audio_path)])\n\nsorted_images = []\n\nfor file_name in file_names:\n    images = filter_images(files, file_name)\n    sorted_images.extend(sort_images(images))\n\ntst_files = L(f'{test_img_path}/{image}' for image in sorted_images)\ntst_files","metadata":{"execution":{"iopub.status.busy":"2023-05-15T22:14:56.232117Z","iopub.execute_input":"2023-05-15T22:14:56.232927Z","iopub.status.idle":"2023-05-15T22:14:56.251424Z","shell.execute_reply.started":"2023-05-15T22:14:56.232871Z","shell.execute_reply":"2023-05-15T22:14:56.250062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = load_learner(path/'trained-clf/convnext_tiny_20_epochs.pkl')\n\n# use multithreading for inference\nlearn.model = learn.model.to('cpu')  \nlearn.dls.device = 'cpu'\nlearn.dls.bs = 1\nlearn.dls.num_workers = 4\ntst_dl = learn.dls.test_dl(tst_files, shuffle=False)\ntst_dl.show_batch()","metadata":{"execution":{"iopub.status.busy":"2023-05-15T22:14:56.253259Z","iopub.execute_input":"2023-05-15T22:14:56.254222Z","iopub.status.idle":"2023-05-15T22:14:59.077678Z","shell.execute_reply.started":"2023-05-15T22:14:56.254175Z","shell.execute_reply":"2023-05-15T22:14:59.076813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs, targets = learn.tta(dl=tst_dl)\nprobs.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-15T22:16:56.701555Z","iopub.execute_input":"2023-05-15T22:16:56.702147Z","iopub.status.idle":"2023-05-15T22:17:16.689581Z","shell.execute_reply.started":"2023-05-15T22:16:56.702096Z","shell.execute_reply":"2023-05-15T22:17:16.688438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(path/'birdclef-2023/sample_submission.csv')\nlist_samp = sample_sub.columns.tolist()[1:]\nprint(list_samp == learn.dls.vocab)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T22:17:20.633409Z","iopub.execute_input":"2023-05-15T22:17:20.634001Z","iopub.status.idle":"2023-05-15T22:17:20.664556Z","shell.execute_reply.started":"2023-05-15T22:17:20.633939Z","shell.execute_reply":"2023-05-15T22:17:20.663277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df = pd.DataFrame(columns=sample_sub.columns)\nsub_df","metadata":{"execution":{"iopub.status.busy":"2023-05-15T22:17:21.823544Z","iopub.execute_input":"2023-05-15T22:17:21.824452Z","iopub.status.idle":"2023-05-15T22:17:21.850138Z","shell.execute_reply.started":"2023-05-15T22:17:21.824399Z","shell.execute_reply":"2023-05-15T22:17:21.849255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, file in enumerate(sorted_images):\n    prob = probs[i].numpy()\n    num_rows = len(prob)\n    row_ids = [file.split('.')[0]]\n    df = pd.DataFrame(columns=sample_sub.columns)\n    \n    df[sample_sub.columns[0]] = row_ids\n    df[sample_sub.columns[1:]] = prob.astype(float)\n    \n    sub_df = pd.concat([sub_df,df]).reset_index(drop=True)\n\nsub_df","metadata":{"execution":{"iopub.status.busy":"2023-05-15T22:17:22.786728Z","iopub.execute_input":"2023-05-15T22:17:22.787585Z","iopub.status.idle":"2023-05-15T22:17:31.929807Z","shell.execute_reply.started":"2023-05-15T22:17:22.787542Z","shell.execute_reply":"2023-05-15T22:17:31.928357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-15T22:16:10.913652Z","iopub.execute_input":"2023-05-15T22:16:10.914439Z","iopub.status.idle":"2023-05-15T22:16:10.999123Z","shell.execute_reply.started":"2023-05-15T22:16:10.914388Z","shell.execute_reply":"2023-05-15T22:16:10.997838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}