{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook, we'll use a pre-trained machine learning model to generate a submission to the [BirdClef2023 competition](https://www.kaggle.com/c/birdclef-2023).  The goal of the competition is to identify Eastern African bird species by sound.","metadata":{}},{"cell_type":"markdown","source":"## Step 1: Imports","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_io as tfio\n\nimport pandas as pd\nimport numpy as np\nimport librosa\nimport glob\n\nimport torch\nimport torchaudio\nimport torch.nn.functional as F\nimport torch.nn as nn\nfrom torchvision.models import resnet50\n\nimport cv2\nimport csv\nimport io\nimport os\n\nfrom IPython.display import Audio","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-04-19T17:09:17.191386Z","iopub.execute_input":"2023-04-19T17:09:17.192678Z","iopub.status.idle":"2023-04-19T17:09:22.095465Z","shell.execute_reply.started":"2023-04-19T17:09:17.192632Z","shell.execute_reply":"2023-04-19T17:09:22.093903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    metadata_path = '/kaggle/input/birdclef-2023/train_metadata.csv'\n    test_dir = '/kaggle/input/birdclef-2023/test_soundscapes/'\n    savemodel_path = '/kaggle/input/baseline/'\n    Experience = '003-try_loss_best.mdl'\n    img_size = 224","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:09:22.102774Z","iopub.execute_input":"2023-04-19T17:09:22.103205Z","iopub.status.idle":"2023-04-19T17:09:22.109389Z","shell.execute_reply.started":"2023-04-19T17:09:22.103160Z","shell.execute_reply":"2023-04-19T17:09:22.108029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ResNetBase(nn.Module):\n    def __init__(self):\n        super(ResNetBase, self).__init__()\n        # down sample 32 + average pooling\n        self.backbone = resnet50()\n        self.backbone.fc = nn.Linear(2048, 264, bias=True)\n\n    def forward(self, x):\n        x = self.backbone(x)\n        # y = torch.nn.functional.softmax(x, dim=1)\n        return x\n\ndef load_model(model, path):\n    params = torch.load(path, map_location=torch.device('cpu'))\n    model.load_state_dict(params)\n    return model","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:09:22.111101Z","iopub.execute_input":"2023-04-19T17:09:22.111907Z","iopub.status.idle":"2023-04-19T17:09:22.123149Z","shell.execute_reply.started":"2023-04-19T17:09:22.111864Z","shell.execute_reply":"2023-04-19T17:09:22.122103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = ResNetBase()\nPATH = os.path.join(CFG.savemodel_path, CFG.Experience)\nmodel = load_model(model, PATH)\nmodel.eval()\nprint(\"Model OK\")","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:09:22.125662Z","iopub.execute_input":"2023-04-19T17:09:22.127105Z","iopub.status.idle":"2023-04-19T17:09:22.739345Z","shell.execute_reply.started":"2023-04-19T17:09:22.127058Z","shell.execute_reply":"2023-04-19T17:09:22.738322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 3: Match the model's output with the bird species in the competition\n\nThe competition includes 264 classes of birds, 261 of which exist in this model. We'll set up a way to map the model's output logits to our competition.","metadata":{}},{"cell_type":"code","source":"# Find the name of the class with the top score when mean-aggregated across frames.\n# def class_names_from_csv(class_map_csv_text):\n#     \"\"\"Returns list of class names corresponding to score vector.\"\"\"\n#     with open(labels_path) as csv_file:\n#         csv_reader = csv.reader(csv_file, delimiter=',')\n#         class_names = [mid for mid, desc in csv_reader]\n#         return class_names[1:]\n\n## note that the bird classifier classifies a much larger set of birds than the\n## competition, so we need to load the model's set of class names or else our \n## indices will be off.","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:09:22.740873Z","iopub.execute_input":"2023-04-19T17:09:22.741248Z","iopub.status.idle":"2023-04-19T17:09:22.746666Z","shell.execute_reply.started":"2023-04-19T17:09:22.741210Z","shell.execute_reply":"2023-04-19T17:09:22.745398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\ntrain_metadata.head()\ncompetition_classes = sorted(train_metadata.primary_label.unique())\n\n# forced_defaults = 0\n# competition_class_map = []\n# for c in competition_classes:\n#     try:\n#         i = classes.index(c)\n#         competition_class_map.append(i)\n#     except:\n#         competition_class_map.append(0)\n#         forced_defaults += 1\n        \n## this is the count of classes not supported by our pretrained model\n## you could choose to simply not predict these, set a default as above,\n## or create your own model using the pretrained model as a base.","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:09:22.748340Z","iopub.execute_input":"2023-04-19T17:09:22.748702Z","iopub.status.idle":"2023-04-19T17:09:22.826692Z","shell.execute_reply.started":"2023-04-19T17:09:22.748667Z","shell.execute_reply":"2023-04-19T17:09:22.825336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 4: Preprocess the data\n\nThe following functions are one way to load the audio provided and break it up into the five-second samples with a sample rate of 32,000 required by the competition.","metadata":{}},{"cell_type":"code","source":"def frame_audio(\n      audio_array: np.ndarray,\n      window_size_s: float = 5.0,\n      hop_size_s: float = 5.0,\n      sample_rate = 32000,\n      ) -> np.ndarray:\n    \n    \"\"\"Helper function for framing audio for inference.\"\"\"\n    \"\"\" using tf.signal \"\"\"\n    if window_size_s is None or window_size_s < 0:\n        return audio_array[np.newaxis, :]\n    frame_length = int(window_size_s * sample_rate)\n    hop_length = int(hop_size_s * sample_rate)\n    framed_audio = tf.signal.frame(audio_array, frame_length, hop_length, pad_end=True)\n    return framed_audio\n\ndef ensure_sample_rate(waveform, original_sample_rate,\n                       desired_sample_rate=32000):\n    \"\"\"Resample waveform if required.\"\"\"\n    if original_sample_rate != desired_sample_rate:\n        waveform = tfio.audio.resample(waveform, original_sample_rate, desired_sample_rate)\n    return desired_sample_rate, waveform","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:09:22.828395Z","iopub.execute_input":"2023-04-19T17:09:22.828754Z","iopub.status.idle":"2023-04-19T17:09:22.838242Z","shell.execute_reply.started":"2023-04-19T17:09:22.828719Z","shell.execute_reply":"2023-04-19T17:09:22.836965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def wave2spec(wave, sr: int) -> torch.Tensor():\n    trans = torchaudio.transforms.MelSpectrogram(sr, n_mels=32)\n    mel_spec = trans(wave).numpy()[0]\n    mel_spec = np.delete(mel_spec, np.argwhere(np.all(mel_spec[..., :] == 0, axis=0)), axis=1)\n    mel_resize = 10 * np.log10(cv2.resize(mel_spec, dsize=(CFG.img_size, CFG.img_size)))\n    m, s = np.mean(mel_resize), np.std(mel_resize)\n    mel_resize = (mel_resize - m) / s\n    mel_tensor = torch.tensor(np.reshape(mel_resize, (1, CFG.img_size, CFG.img_size)))\n    mel_tensor = torch.cat([mel_tensor, mel_tensor, mel_tensor], dim=0)\n    return mel_tensor\n","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:09:22.839330Z","iopub.execute_input":"2023-04-19T17:09:22.839730Z","iopub.status.idle":"2023-04-19T17:09:22.851323Z","shell.execute_reply.started":"2023-04-19T17:09:22.839694Z","shell.execute_reply":"2023-04-19T17:09:22.849713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 5: Make predictions\n\nEach test sample is cut into 5-second chunks. We use the pretrained model to return probabilities for all 10k birds included in the model, then pull out the classes used in this competition to create a final submission row. Note that we are NOT doing anything special to handle the 3 missing classes; those will need fine-tuning / transfer learning, which will be handled in a separate notebook.","metadata":{}},{"cell_type":"code","source":"def predict_for_sample(filename, pred, frame_limit_secs=None):\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    \n    audio, sample_rate = librosa.load(filename)\n    sample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\n    \n    fixed_tm = frame_audio(wav_data).numpy() # 120 x 160000\n    \n    frame = 5\n    all_logits = model(torch.unsqueeze(wave2spec(torch.tensor(fixed_tm[:1]), sample_rate), 0)).detach().numpy()\n\n    for window in fixed_tm[1:]:\n        if frame_limit_secs and frame > frame_limit_secs:\n            continue\n        \n        logits = model(torch.unsqueeze(wave2spec(torch.tensor(window[np.newaxis, :]), sample_rate), 0)).detach().numpy()\n        all_logits = np.concatenate([all_logits, logits], axis=0)\n        frame += 5\n\n    frame = 5\n    all_probabilities = []\n    for frame_logits in all_logits:\n        pred['row_id'].append(file_id + \"_\" + str(frame))\n        probabilities = tf.nn.softmax(frame_logits).numpy().reshape(264)\n        for j, bird in enumerate(competition_classes):\n            pred[bird].append(probabilities[j])\n        ## set the appropriate row in the sample submission\n        # sample_submission.loc[sample_submission.row_id == file_id + \"_\" + str(frame), competition_classes] = probabilities[competition_class_map]\n        frame += 5\n    return pred","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:09:22.853017Z","iopub.execute_input":"2023-04-19T17:09:22.853673Z","iopub.status.idle":"2023-04-19T17:09:22.868467Z","shell.execute_reply.started":"2023-04-19T17:09:22.853629Z","shell.execute_reply":"2023-04-19T17:09:22.867133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 6: Generate a submission\n\nNow we process all of the test samples as discussed above, creating output rows, and saving them in the provided `sample_submission.csv`. Finally, we save these rows to our final output file: `submission.csv`. This is the file that gets submitted and scored when you submit the notebook.","metadata":{}},{"cell_type":"code","source":"test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\ntest_samples","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:09:22.869898Z","iopub.execute_input":"2023-04-19T17:09:22.870473Z","iopub.status.idle":"2023-04-19T17:09:22.884700Z","shell.execute_reply.started":"2023-04-19T17:09:22.870427Z","shell.execute_reply":"2023-04-19T17:09:22.883455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\nsample_sub[competition_classes] = sample_sub[competition_classes].astype(np.float32)\nsample_sub = sample_sub[0:0]\nsample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:09:22.886545Z","iopub.execute_input":"2023-04-19T17:09:22.886981Z","iopub.status.idle":"2023-04-19T17:09:22.974215Z","shell.execute_reply.started":"2023-04-19T17:09:22.886928Z","shell.execute_reply":"2023-04-19T17:09:22.973059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = {'row_id': []}\nfor species_code in competition_classes:\n    pred[species_code] = []\nfor sample_filename in test_samples:\n    pred = predict_for_sample(sample_filename, pred)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:09:22.975483Z","iopub.execute_input":"2023-04-19T17:09:22.975845Z","iopub.status.idle":"2023-04-19T17:09:43.234524Z","shell.execute_reply.started":"2023-04-19T17:09:22.975811Z","shell.execute_reply":"2023-04-19T17:09:43.232753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = pd.DataFrame(pred, columns = ['row_id'] + competition_classes)\nresults","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:09:48.023338Z","iopub.execute_input":"2023-04-19T17:09:48.024026Z","iopub.status.idle":"2023-04-19T17:09:48.087760Z","shell.execute_reply.started":"2023-04-19T17:09:48.023938Z","shell.execute_reply":"2023-04-19T17:09:48.086230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-04-19T17:09:43.286836Z","iopub.execute_input":"2023-04-19T17:09:43.287360Z","iopub.status.idle":"2023-04-19T17:09:43.334911Z","shell.execute_reply.started":"2023-04-19T17:09:43.287307Z","shell.execute_reply":"2023-04-19T17:09:43.333481Z"},"trusted":true},"execution_count":null,"outputs":[]}]}