{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":44224,"databundleVersionId":5188730,"sourceType":"competition"},{"sourceId":5183470,"sourceType":"datasetVersion","datasetId":3013640},{"sourceId":3836,"sourceType":"modelInstanceVersion","modelInstanceId":2739,"modelId":319}],"dockerImageVersionId":30407,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook, we'll use a pre-trained machine learning model to generate a submission to the [BirdClef2023 competition](https://www.kaggle.com/c/birdclef-2023).  The goal of the competition is to identify Eastern African bird species by sound.","metadata":{}},{"cell_type":"markdown","source":"## Step 1: Imports","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_io as tfio\n\nimport pandas as pd\nimport numpy as np\nimport librosa\nimport glob\n\nimport csv\nimport io\n\nfrom IPython.display import Audio","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-21T05:18:21.480269Z","iopub.execute_input":"2023-03-21T05:18:21.481084Z","iopub.status.idle":"2023-03-21T05:18:21.488424Z","shell.execute_reply.started":"2023-03-21T05:18:21.481020Z","shell.execute_reply":"2023-03-21T05:18:21.486955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 2: Explore the training data\n\nWe'll start by loading a couple of training examples and using the IPython.display.Audio module to play them!","metadata":{}},{"cell_type":"code","source":"# Load a sample audio files from two different species\naudio_abe, sr_abe = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/abethr1/XC128013.ogg\")\naudio_abh, sr_abh = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/abhori1/XC127317.ogg\")","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:18:27.361984Z","iopub.execute_input":"2023-03-21T05:18:27.363280Z","iopub.status.idle":"2023-03-21T05:18:40.074670Z","shell.execute_reply.started":"2023-03-21T05:18:27.363207Z","shell.execute_reply":"2023-03-21T05:18:40.073320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_abe.shape, sr_abe","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:18:40.077297Z","iopub.execute_input":"2023-03-21T05:18:40.078374Z","iopub.status.idle":"2023-03-21T05:18:40.087767Z","shell.execute_reply.started":"2023-03-21T05:18:40.078316Z","shell.execute_reply":"2023-03-21T05:18:40.086389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_abe.min(), audio_abe.max()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:18:47.046427Z","iopub.execute_input":"2023-03-21T05:18:47.047632Z","iopub.status.idle":"2023-03-21T05:18:47.055205Z","shell.execute_reply.started":"2023-03-21T05:18:47.047583Z","shell.execute_reply":"2023-03-21T05:18:47.054257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_abh.shape, sr_abh","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:19:05.701648Z","iopub.execute_input":"2023-03-21T05:19:05.702061Z","iopub.status.idle":"2023-03-21T05:19:05.709404Z","shell.execute_reply.started":"2023-03-21T05:19:05.702011Z","shell.execute_reply":"2023-03-21T05:19:05.708214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"1005696 // 22050","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:19:22.976706Z","iopub.execute_input":"2023-03-21T05:19:22.977175Z","iopub.status.idle":"2023-03-21T05:19:22.985309Z","shell.execute_reply.started":"2023-03-21T05:19:22.977114Z","shell.execute_reply":"2023-03-21T05:19:22.984111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abe, rate=sr_abe)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:19:08.636079Z","iopub.execute_input":"2023-03-21T05:19:08.636470Z","iopub.status.idle":"2023-03-21T05:19:08.711378Z","shell.execute_reply.started":"2023-03-21T05:19:08.636436Z","shell.execute_reply":"2023-03-21T05:19:08.710388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"975581 // 22050","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:19:33.952693Z","iopub.execute_input":"2023-03-21T05:19:33.953948Z","iopub.status.idle":"2023-03-21T05:19:33.960847Z","shell.execute_reply.started":"2023-03-21T05:19:33.953886Z","shell.execute_reply":"2023-03-21T05:19:33.959640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abh, rate=sr_abh)","metadata":{"execution":{"iopub.status.busy":"2023-03-19T09:32:03.582864Z","iopub.execute_input":"2023-03-19T09:32:03.584243Z","iopub.status.idle":"2023-03-19T09:32:03.653037Z","shell.execute_reply.started":"2023-03-19T09:32:03.584160Z","shell.execute_reply":"2023-03-19T09:32:03.651341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sr_abh","metadata":{"execution":{"iopub.status.busy":"2023-03-19T09:32:13.079720Z","iopub.execute_input":"2023-03-19T09:32:13.080187Z","iopub.status.idle":"2023-03-19T09:32:13.087896Z","shell.execute_reply.started":"2023-03-19T09:32:13.080129Z","shell.execute_reply":"2023-03-19T09:32:13.086593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 3: Match the model's output with the bird species in the competition\n\nThe competition includes 264 classes of birds, 261 of which exist in this model. We'll set up a way to map the model's output logits to our competition.","metadata":{}},{"cell_type":"code","source":"model = hub.load('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1')\nlabels_path = hub.resolve('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1') + \"/assets/label.csv\"","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:19:42.546963Z","iopub.execute_input":"2023-03-21T05:19:42.547425Z","iopub.status.idle":"2023-03-21T05:19:51.213378Z","shell.execute_reply.started":"2023-03-21T05:19:42.547383Z","shell.execute_reply":"2023-03-21T05:19:51.212225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_2 = tf.keras.models.load_model(\"/kaggle/input/bird-clef/model.h5\")\nmodel_2","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:20:31.364956Z","iopub.execute_input":"2023-03-21T05:20:31.365411Z","iopub.status.idle":"2023-03-21T05:20:40.928194Z","shell.execute_reply.started":"2023-03-21T05:20:31.365372Z","shell.execute_reply":"2023-03-21T05:20:40.926909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find the name of the class with the top score when mean-aggregated across frames.\ndef class_names_from_csv(class_map_csv_text):\n    \"\"\"Returns list of class names corresponding to score vector.\"\"\"\n    with open(labels_path) as csv_file:\n        csv_reader = csv.reader(csv_file, delimiter=',')\n        class_names = [mid for mid, desc in csv_reader]\n        return class_names[1:]\n\n## note that the bird classifier classifies a much larger set of birds than the\n## competition, so we need to load the model's set of class names or else our \n## indices will be off.\nclasses = class_names_from_csv(labels_path)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:20:46.890388Z","iopub.execute_input":"2023-03-21T05:20:46.890814Z","iopub.status.idle":"2023-03-21T05:20:46.914293Z","shell.execute_reply.started":"2023-03-21T05:20:46.890773Z","shell.execute_reply":"2023-03-21T05:20:46.913108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\ntrain_metadata.head()\ncompetition_classes = sorted(train_metadata.primary_label.unique())\n\nforced_defaults = 0\ncompetition_class_map = []\nfor c in competition_classes:\n    try:\n        i = classes.index(c)\n        competition_class_map.append(i)\n    except:\n        competition_class_map.append(0)\n        forced_defaults += 1\n        \n## this is the count of classes not supported by our pretrained model\n## you could choose to simply not predict these, set a default as above,\n## or create your own model using the pretrained model as a base.\nforced_defaults","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:20:54.539977Z","iopub.execute_input":"2023-03-21T05:20:54.540742Z","iopub.status.idle":"2023-03-21T05:20:54.719482Z","shell.execute_reply.started":"2023-03-21T05:20:54.540698Z","shell.execute_reply":"2023-03-21T05:20:54.718153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:20:57.869277Z","iopub.execute_input":"2023-03-21T05:20:57.869733Z","iopub.status.idle":"2023-03-21T05:20:57.902394Z","shell.execute_reply.started":"2023-03-21T05:20:57.869685Z","shell.execute_reply":"2023-03-21T05:20:57.900701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_class_map","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:21:01.275846Z","iopub.execute_input":"2023-03-21T05:21:01.277147Z","iopub.status.idle":"2023-03-21T05:21:01.289424Z","shell.execute_reply.started":"2023-03-21T05:21:01.277095Z","shell.execute_reply":"2023-03-21T05:21:01.287953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 4: Preprocess the data\n\nThe following functions are one way to load the audio provided and break it up into the five-second samples with a sample rate of 32,000 required by the competition.","metadata":{}},{"cell_type":"code","source":"def frame_audio(\n      audio_array: np.ndarray,\n      window_size_s: float = 5.0,\n      hop_size_s: float = 5.0,\n      sample_rate = 32000,\n      ) -> np.ndarray:\n    \n    \"\"\"Helper function for framing audio for inference.\"\"\"\n    \"\"\" using tf.signal \"\"\"\n    if window_size_s is None or window_size_s < 0:\n        return audio_array[np.newaxis, :]\n    frame_length = int(window_size_s * sample_rate)\n    hop_length = int(hop_size_s * sample_rate)\n    framed_audio = tf.signal.frame(audio_array, frame_length, hop_length, pad_end=True)\n    return framed_audio\n\ndef ensure_sample_rate(waveform, original_sample_rate,\n                       desired_sample_rate=32000):\n    \"\"\"Resample waveform if required.\"\"\"\n    if original_sample_rate != desired_sample_rate:\n        waveform = tfio.audio.resample(waveform, original_sample_rate, desired_sample_rate)\n    return desired_sample_rate, waveform","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:21:08.481659Z","iopub.execute_input":"2023-03-21T05:21:08.482275Z","iopub.status.idle":"2023-03-21T05:21:08.491387Z","shell.execute_reply.started":"2023-03-21T05:21:08.482231Z","shell.execute_reply":"2023-03-21T05:21:08.490112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Below we load one training sample - use the Audio function to listen to the samples inside the notebook!","metadata":{}},{"cell_type":"code","source":"audio, sample_rate = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/afghor1/XC156639.ogg\")\nsample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\nAudio(wav_data, rate=sample_rate)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:21:12.409491Z","iopub.execute_input":"2023-03-21T05:21:12.409897Z","iopub.status.idle":"2023-03-21T05:21:13.584349Z","shell.execute_reply.started":"2023-03-21T05:21:12.409861Z","shell.execute_reply":"2023-03-21T05:21:13.583408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 5: Make predictions\n\nEach test sample is cut into 5-second chunks. We use the pretrained model to return probabilities for all 10k birds included in the model, then pull out the classes used in this competition to create a final submission row. Note that we are NOT doing anything special to handle the 3 missing classes; those will need fine-tuning / transfer learning, which will be handled in a separate notebook.","metadata":{}},{"cell_type":"code","source":"fixed_tm = frame_audio(wav_data)\nlogits, embeddings = model.infer_tf(fixed_tm[:1])\nprobabilities = tf.nn.softmax(logits)\nargmax = np.argmax(probabilities)\nprint(f\"The audio is from the class {classes[argmax]} (element:{argmax} in the label.csv file), with probability of {probabilities[0][argmax]}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:21:36.550808Z","iopub.execute_input":"2023-03-21T05:21:36.551248Z","iopub.status.idle":"2023-03-21T05:21:47.091917Z","shell.execute_reply.started":"2023-03-21T05:21:36.551205Z","shell.execute_reply":"2023-03-21T05:21:47.091022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_for_sample(filename, sample_submission, frame_limit_secs=None):\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    \n    audio, sample_rate = librosa.load(filename)\n    sample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\n    \n    fixed_tm = frame_audio(wav_data)\n    \n    frame = 5\n    all_logits, all_embeddings = model.infer_tf(fixed_tm[:1])\n    for window in fixed_tm[1:]:\n        if frame_limit_secs and frame > frame_limit_secs:\n            continue\n        \n        logits, embeddings = model.infer_tf(window[np.newaxis, :])\n        all_logits = np.concatenate([all_logits, logits], axis=0)\n        frame += 5\n    \n    frame = 5\n    all_probabilities = []\n    for frame_logits in all_logits:\n        probabilities = tf.nn.softmax(frame_logits).numpy()\n        \n        ## set the appropriate row in the sample submission\n        sample_submission.loc[sample_submission.row_id == file_id + \"_\" + str(frame), competition_classes] = probabilities[competition_class_map]\n        frame += 5","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:21:49.938759Z","iopub.execute_input":"2023-03-21T05:21:49.939743Z","iopub.status.idle":"2023-03-21T05:21:49.952588Z","shell.execute_reply.started":"2023-03-21T05:21:49.939697Z","shell.execute_reply":"2023-03-21T05:21:49.950799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 6: Generate a submission\n\nNow we process all of the test samples as discussed above, creating output rows, and saving them in the provided `sample_submission.csv`. Finally, we save these rows to our final output file: `submission.csv`. This is the file that gets submitted and scored when you submit the notebook.","metadata":{}},{"cell_type":"code","source":"def preprocess_test(audio_tensor):\n    tensor = tf.cast(audio_tensor, tf.float32) / 32768.0\n    spectrogram = tfio.audio.spectrogram(tensor, nfft=512, window=512, stride=256)\n    spectrogram = tfio.audio.dbscale(spectrogram, top_db=80)\n    spectrogram = tf.expand_dims(spectrogram, axis=-1)\n    spectrogram = tf.image.resize(spectrogram, (256, 256))\n    spectrogram = (spectrogram - tf.reduce_min(spectrogram)) / (tf.reduce_max(spectrogram) - tf.reduce_min(spectrogram)) * 255.0\n    return spectrogram","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:29:25.542222Z","iopub.execute_input":"2023-03-21T05:29:25.542649Z","iopub.status.idle":"2023-03-21T05:29:25.551425Z","shell.execute_reply.started":"2023-03-21T05:29:25.542608Z","shell.execute_reply":"2023-03-21T05:29:25.550071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_2","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:27:59.210895Z","iopub.execute_input":"2023-03-21T05:27:59.211349Z","iopub.status.idle":"2023-03-21T05:27:59.219338Z","shell.execute_reply.started":"2023-03-21T05:27:59.211307Z","shell.execute_reply":"2023-03-21T05:27:59.218097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_inference(tensor):\n    image = preprocess_test(tensor)\n    print(image.shape)\n    return model_2.predict(image)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:28:05.148146Z","iopub.execute_input":"2023-03-21T05:28:05.148629Z","iopub.status.idle":"2023-03-21T05:28:05.158511Z","shell.execute_reply.started":"2023-03-21T05:28:05.148588Z","shell.execute_reply":"2023-03-21T05:28:05.157307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\ntest_samples","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:22:12.969308Z","iopub.execute_input":"2023-03-21T05:22:12.969729Z","iopub.status.idle":"2023-03-21T05:22:12.979794Z","shell.execute_reply.started":"2023-03-21T05:22:12.969692Z","shell.execute_reply":"2023-03-21T05:22:12.978435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\nsample_sub[competition_classes] = sample_sub[competition_classes].astype(np.float32)\nsample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:22:16.042662Z","iopub.execute_input":"2023-03-21T05:22:16.043091Z","iopub.status.idle":"2023-03-21T05:22:16.145141Z","shell.execute_reply.started":"2023-03-21T05:22:16.043051Z","shell.execute_reply":"2023-03-21T05:22:16.144154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_for_sample_v2(filename, sample_submission, frame_limit_secs=None):\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    audio, sample_rate = librosa.load(filename)\n    sample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\n    fixed_tm = frame_audio(wav_data)\n    frame = 5\n    all_logits = make_inference(fixed_tm[:1])\n    for window in fixed_tm[1:]:\n        if frame_limit_secs and frame > frame_limit_secs:\n            continue\n        logits = make_inference(window[np.newaxis, :])\n        all_logits = np.concatenate([all_logits, logits], axis=0)\n        frame += 5\n    frame = 5\n    all_probabilities = []\n    for frame_logits in all_logits:\n        probabilities = tf.nn.softmax(frame_logits).numpy()\n        ## set the appropriate row in the sample submission\n        sample_submission.loc[sample_submission.row_id == file_id + \"_\" + str(frame), competition_classes] = probabilities\n        frame += 5","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:32:28.842516Z","iopub.execute_input":"2023-03-21T05:32:28.842941Z","iopub.status.idle":"2023-03-21T05:32:28.852022Z","shell.execute_reply.started":"2023-03-21T05:32:28.842902Z","shell.execute_reply":"2023-03-21T05:32:28.850688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio, sample_rate = librosa.load(\"/kaggle/input/birdclef-2023/test_soundscapes/soundscape_29201.ogg\")\nsample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\nAudio(wav_data, rate=sample_rate)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:22:22.649816Z","iopub.execute_input":"2023-03-21T05:22:22.650233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frame_limit_secs = 15 if sample_sub.shape[0] == 3 else None\nfor sample_filename in test_samples:\n    predict_for_sample_v2(sample_filename, sample_sub, frame_limit_secs=15)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:32:32.282859Z","iopub.execute_input":"2023-03-21T05:32:32.283832Z","iopub.status.idle":"2023-03-21T05:32:36.813362Z","shell.execute_reply.started":"2023-03-21T05:32:32.283785Z","shell.execute_reply":"2023-03-21T05:32:36.812095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T05:32:49.965948Z","iopub.execute_input":"2023-03-21T05:32:49.966398Z","iopub.status.idle":"2023-03-21T05:32:49.987274Z","shell.execute_reply.started":"2023-03-21T05:32:49.966341Z","shell.execute_reply":"2023-03-21T05:32:49.985953Z"},"trusted":true},"execution_count":null,"outputs":[]}]}