{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook, we'll use a pre-trained machine learning model to generate a submission to the [BirdClef2023 competition](https://www.kaggle.com/c/birdclef-2023).  The goal of the competition is to identify Eastern African bird species by sound.","metadata":{"_uuid":"e08d4e7d-0f8f-40c4-9d3b-b141c1a8eb83","_cell_guid":"00ced965-54b9-4393-9503-a9b2d520421e","trusted":true}},{"cell_type":"markdown","source":"## Step 1: Imports","metadata":{"_uuid":"4d556ee7-ec46-402b-a470-f9b20f619d85","_cell_guid":"83a2f117-6041-4605-9d78-c9ca25c8d930","trusted":true}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_io as tfio\n\nimport pandas as pd\nimport numpy as np\nimport librosa\nimport glob\n\nimport csv\nimport io\n\nfrom IPython.display import Audio","metadata":{"_uuid":"ab485b37-de40-4cff-bbc6-ade3857e5a9a","_cell_guid":"83ab1b30-1652-4915-9eaf-9441ff2c2a30","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:26:05.893236Z","iopub.execute_input":"2023-03-07T18:26:05.893687Z","iopub.status.idle":"2023-03-07T18:26:15.443097Z","shell.execute_reply.started":"2023-03-07T18:26:05.89362Z","shell.execute_reply":"2023-03-07T18:26:15.441763Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 2: Explore the training data\n\nWe'll start by loading a couple of training examples and using the IPython.display.Audio module to play them!","metadata":{"_uuid":"2f360fe9-fded-4aea-9367-9abd8a7dd5b4","_cell_guid":"99f17937-91b6-438c-a775-dd682e1a0f8b","trusted":true}},{"cell_type":"code","source":"# Load a sample audio files from two different species\naudio_abe, sr_abe = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/abethr1/XC128013.ogg\")\naudio_abh, sr_abh = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/abhori1/XC127317.ogg\")","metadata":{"_uuid":"480b72d7-cf1c-4437-a579-b4326dfe79a2","_cell_guid":"a97b6da7-5510-4c52-b07b-fbcdf29902c9","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:26:15.44478Z","iopub.execute_input":"2023-03-07T18:26:15.445668Z","iopub.status.idle":"2023-03-07T18:26:25.821183Z","shell.execute_reply.started":"2023-03-07T18:26:15.445601Z","shell.execute_reply":"2023-03-07T18:26:25.819673Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abe, rate=sr_abe)","metadata":{"_uuid":"d8de14b5-b0f7-4af4-8d96-6d0c39d7891a","_cell_guid":"ea0e687e-8ddb-4f42-b8ca-93d1a9dd67d2","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:26:25.822527Z","iopub.execute_input":"2023-03-07T18:26:25.823301Z","iopub.status.idle":"2023-03-07T18:26:25.873528Z","shell.execute_reply.started":"2023-03-07T18:26:25.823267Z","shell.execute_reply":"2023-03-07T18:26:25.872446Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abh, rate=sr_abh)","metadata":{"_uuid":"b83954b1-c231-41d3-ba4f-31e03a962150","_cell_guid":"5b9bbc84-60fc-4898-854a-387d8c39f252","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:26:25.874393Z","iopub.execute_input":"2023-03-07T18:26:25.874881Z","iopub.status.idle":"2023-03-07T18:26:25.915097Z","shell.execute_reply.started":"2023-03-07T18:26:25.874851Z","shell.execute_reply":"2023-03-07T18:26:25.914013Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 3: Match the model's output with the bird species in the competition\n\nThe competition includes 264 classes of birds, 261 of which exist in this model. We'll set up a way to map the model's output logits to our competition.","metadata":{"_uuid":"93395607-b842-44ca-9565-dc62bacbc68e","_cell_guid":"1ec04201-f3f8-461b-a34a-a1b296483ae4","trusted":true}},{"cell_type":"code","source":"model = hub.load('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1')\nlabels_path = hub.resolve('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1') + \"/assets/label.csv\"","metadata":{"_uuid":"c18845cf-f407-459b-8bb9-dea67f9f2bfe","_cell_guid":"a976e1de-1da8-4ae0-80da-b744b57be339","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:26:25.918763Z","iopub.execute_input":"2023-03-07T18:26:25.919053Z","iopub.status.idle":"2023-03-07T18:26:33.264685Z","shell.execute_reply.started":"2023-03-07T18:26:25.919023Z","shell.execute_reply":"2023-03-07T18:26:33.263005Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find the name of the class with the top score when mean-aggregated across frames.\ndef class_names_from_csv(class_map_csv_text):\n    \"\"\"Returns list of class names corresponding to score vector.\"\"\"\n    with open(labels_path) as csv_file:\n        csv_reader = csv.reader(csv_file, delimiter=',')\n        class_names = [mid for mid, desc in csv_reader]\n        return class_names[1:]\n\n## note that the bird classifier classifies a much larger set of birds than the\n## competition, so we need to load the model's set of class names or else our \n## indices will be off.\nclasses = class_names_from_csv(labels_path)","metadata":{"_uuid":"90d660ba-5a50-4d68-8a19-970a3701a175","_cell_guid":"01bd22ef-87e3-4b9b-b7fb-bbc969d6235e","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:26:33.266455Z","iopub.execute_input":"2023-03-07T18:26:33.266825Z","iopub.status.idle":"2023-03-07T18:26:33.290684Z","shell.execute_reply.started":"2023-03-07T18:26:33.266789Z","shell.execute_reply":"2023-03-07T18:26:33.289205Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\ntrain_metadata.head()\ncompetition_classes = sorted(train_metadata.primary_label.unique())\n\nforced_defaults = 0\ncompetition_class_map = []\nfor c in competition_classes:\n    try:\n        i = classes.index(c)\n        competition_class_map.append(i)\n    except:\n        competition_class_map.append(0)\n        forced_defaults += 1\n        \n## this is the count of classes not supported by our pretrained model\n## you could choose to simply not predict these, set a default as above,\n## or create your own model using the pretrained model as a base.\nforced_defaults","metadata":{"_uuid":"bd609a2d-f0ce-4382-a66b-f0c96c8996eb","_cell_guid":"b0d29481-5662-4651-93f6-b4ead899ed25","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:26:33.292249Z","iopub.execute_input":"2023-03-07T18:26:33.292678Z","iopub.status.idle":"2023-03-07T18:26:33.458124Z","shell.execute_reply.started":"2023-03-07T18:26:33.292611Z","shell.execute_reply":"2023-03-07T18:26:33.457Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 4: Preprocess the data\n\nThe following functions are one way to load the audio provided and break it up into the five-second samples with a sample rate of 32,000 required by the competition.","metadata":{"_uuid":"1f278365-722f-47bb-87be-66783ecb76b3","_cell_guid":"74d10394-4723-4e57-b708-03e35699c4d4","trusted":true}},{"cell_type":"code","source":"def frame_audio(\n      audio_array: np.ndarray,\n      window_size_s: float = 5.0,\n      hop_size_s: float = 5.0,\n      sample_rate = 32000,\n      ) -> np.ndarray:\n    \n    \"\"\"Helper function for framing audio for inference.\"\"\"\n    \"\"\" using tf.signal \"\"\"\n    if window_size_s is None or window_size_s < 0:\n        return audio_array[np.newaxis, :]\n    frame_length = int(window_size_s * sample_rate)\n    hop_length = int(hop_size_s * sample_rate)\n    framed_audio = tf.signal.frame(audio_array, frame_length, hop_length, pad_end=True)\n    return framed_audio\n\ndef ensure_sample_rate(waveform, original_sample_rate,\n                       desired_sample_rate=32000):\n    \"\"\"Resample waveform if required.\"\"\"\n    if original_sample_rate != desired_sample_rate:\n        waveform = tfio.audio.resample(waveform, original_sample_rate, desired_sample_rate)\n    return desired_sample_rate, waveform","metadata":{"_uuid":"48b17e61-d193-434a-bd8d-16851a3945a4","_cell_guid":"bfe0da59-9ef1-47b0-bcf0-f9ae456ec89b","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:26:33.459554Z","iopub.execute_input":"2023-03-07T18:26:33.45989Z","iopub.status.idle":"2023-03-07T18:26:33.469971Z","shell.execute_reply.started":"2023-03-07T18:26:33.459858Z","shell.execute_reply":"2023-03-07T18:26:33.46833Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Below we load one training sample - use the Audio function to listen to the samples inside the notebook!","metadata":{"_uuid":"46875845-2603-4b70-a26f-beed9e869500","_cell_guid":"5c1f2529-04ac-4888-8557-b77f8a2ac9c7","trusted":true}},{"cell_type":"code","source":"audio, sample_rate = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/afghor1/XC156639.ogg\")\nsample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\nAudio(wav_data, rate=sample_rate)","metadata":{"_uuid":"c3bc120d-e20f-4306-ba19-97ff80cb0f6b","_cell_guid":"56c033a1-eb84-4f5d-9c81-c291338246a1","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:26:33.474189Z","iopub.execute_input":"2023-03-07T18:26:33.474536Z","iopub.status.idle":"2023-03-07T18:26:34.056941Z","shell.execute_reply.started":"2023-03-07T18:26:33.474501Z","shell.execute_reply":"2023-03-07T18:26:34.055418Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 5: Make predictions\n\nEach test sample is cut into 5-second chunks. We use the pretrained model to return probabilities for all 10k birds included in the model, then pull out the classes used in this competition to create a final submission row. Note that we are NOT doing anything special to handle the 3 missing classes; those will need fine-tuning / transfer learning, which will be handled in a separate notebook.","metadata":{"_uuid":"c5f3a77b-b1dd-4cbe-9527-c16dac488dbb","_cell_guid":"d5c8f409-ed0e-42ff-86c4-6338708f7b94","trusted":true}},{"cell_type":"code","source":"fixed_tm = frame_audio(wav_data)\nlogits, embeddings = model.infer_tf(fixed_tm[:1])\nprobabilities = tf.nn.softmax(logits)\nargmax = np.argmax(probabilities)\nprint(f\"The audio is from the class {classes[argmax]} (element:{argmax} in the label.csv file), with probability of {probabilities[0][argmax]}\")","metadata":{"_uuid":"11218c7b-6d20-473a-a181-26b3fe7ff403","_cell_guid":"add6608d-e02c-4bbb-a57e-87f0966806db","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:27:33.85177Z","iopub.execute_input":"2023-03-07T18:27:33.852139Z","iopub.status.idle":"2023-03-07T18:27:42.452689Z","shell.execute_reply.started":"2023-03-07T18:27:33.852108Z","shell.execute_reply":"2023-03-07T18:27:42.451217Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_for_sample(filename, sample_submission, frame_limit_secs=None):\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    \n    audio, sample_rate = librosa.load(filename)\n    sample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\n    \n    fixed_tm = frame_audio(wav_data)\n    \n    frame = 5\n    all_logits, all_embeddings = model.infer_tf(fixed_tm[:1])\n    for window in fixed_tm[1:]:\n        if frame_limit_secs and frame > frame_limit_secs:\n            continue\n        \n        logits, embeddings = model.infer_tf(window[np.newaxis, :])\n        all_logits = np.concatenate([all_logits, logits], axis=0)\n        frame += 5\n    \n    frame = 5\n    all_probabilities = []\n    for frame_logits in all_logits:\n        probabilities = tf.nn.softmax(frame_logits).numpy()\n        \n        ## set the appropriate row in the sample submission\n        sample_submission.loc[sample_submission.row_id == file_id + \"_\" + str(frame), competition_classes] = probabilities[competition_class_map]\n        frame += 5","metadata":{"_uuid":"e73a9528-d9e2-4c72-97bc-a7c6a12a5fed","_cell_guid":"1fee1176-9c27-4eae-8a8f-bdb8e098a52e","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:27:42.456508Z","iopub.execute_input":"2023-03-07T18:27:42.456847Z","iopub.status.idle":"2023-03-07T18:27:42.466058Z","shell.execute_reply.started":"2023-03-07T18:27:42.456815Z","shell.execute_reply":"2023-03-07T18:27:42.464463Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 6: Generate a submission\n\nNow we process all of the test samples as discussed above, creating output rows, and saving them in the provided `sample_submission.csv`. Finally, we save these rows to our final output file: `submission.csv`. This is the file that gets submitted and scored when you submit the notebook.","metadata":{"_uuid":"50988f70-6ea9-44c8-882b-1cb5a5f8d48a","_cell_guid":"1a721a89-1efe-442a-b9b8-27a4b90c9d25","trusted":true}},{"cell_type":"code","source":"test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\ntest_samples","metadata":{"_uuid":"8079776b-563d-430d-9312-ce95d8488ff3","_cell_guid":"ce060650-2aa0-40b8-bb62-c5c8b5dbd0b8","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:27:42.46746Z","iopub.execute_input":"2023-03-07T18:27:42.468572Z","iopub.status.idle":"2023-03-07T18:27:42.491495Z","shell.execute_reply.started":"2023-03-07T18:27:42.46851Z","shell.execute_reply":"2023-03-07T18:27:42.49023Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\nsample_sub[competition_classes] = sample_sub[competition_classes].astype(np.float32)\nsample_sub.head()","metadata":{"_uuid":"2e3551a5-4342-425c-8155-320c468d5f6f","_cell_guid":"b43bda9e-0a40-4f58-bd31-41e100b796f3","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:27:42.493996Z","iopub.execute_input":"2023-03-07T18:27:42.494288Z","iopub.status.idle":"2023-03-07T18:27:42.598852Z","shell.execute_reply.started":"2023-03-07T18:27:42.494253Z","shell.execute_reply":"2023-03-07T18:27:42.597816Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frame_limit_secs = 15 if sample_sub.shape[0] == 3 else None\nfor sample_filename in test_samples:\n    predict_for_sample(sample_filename, sample_sub, frame_limit_secs=15)","metadata":{"_uuid":"25f0842b-d8f3-44aa-980e-cfb1646f58bd","_cell_guid":"6be4fa6d-b838-403b-82a1-e6c1582071cd","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:27:42.600544Z","iopub.execute_input":"2023-03-07T18:27:42.600888Z","iopub.status.idle":"2023-03-07T18:27:48.722475Z","shell.execute_reply.started":"2023-03-07T18:27:42.600855Z","shell.execute_reply":"2023-03-07T18:27:48.720492Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub","metadata":{"_uuid":"d36b3ceb-5f2a-412b-9052-2a089cb7fe6a","_cell_guid":"f057f902-c1cc-428a-bb2a-0164adfa7e97","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:27:48.723696Z","iopub.execute_input":"2023-03-07T18:27:48.724021Z","iopub.status.idle":"2023-03-07T18:27:48.753783Z","shell.execute_reply.started":"2023-03-07T18:27:48.723987Z","shell.execute_reply":"2023-03-07T18:27:48.752703Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.to_csv(\"submission.csv\", index=False)","metadata":{"_uuid":"63a41790-9081-47ae-a79c-2767d21628cc","_cell_guid":"734e6e01-1f5c-4a60-9f1c-6cc7ec6d27c3","collapsed":false,"execution":{"iopub.status.busy":"2023-03-07T18:27:48.755372Z","iopub.execute_input":"2023-03-07T18:27:48.755961Z","iopub.status.idle":"2023-03-07T18:27:48.77547Z","shell.execute_reply.started":"2023-03-07T18:27:48.755926Z","shell.execute_reply":"2023-03-07T18:27:48.774069Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"_uuid":"7232a97c-8fdd-49bf-93cd-d4c83a09160a","_cell_guid":"f16f8349-073e-45c1-be56-6cd54b1ebe3f","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 4: Define a function to extract features from audio clips\n\nThe model we're using requires that we extract Mel spectrograms from our audio files. We'll define a function that extracts these features from a given audio file using the `librosa` library.","metadata":{"_uuid":"f98db6f4-2eb1-4312-8c33-0bdbb395180e","_cell_guid":"272974dc-9c49-4f01-8cf0-2dc51e5f2123","trusted":true}},{"cell_type":"code","source":"def extract_features(file_path):\n    # Load audio file\n    audio, sample_rate = librosa.load(file_path, res_type='kaiser_fast')\n    \n    # Generate mel spectrogram\n    mel_spectrogram = librosa.feature.melspectrogram(y=audio, sr=sample_rate, n_mels=128)\n    mel_spectrogram = librosa.power_to_db(mel_spectrogram, ref=np.max)\n    \n    return mel_spectrogram","metadata":{"_uuid":"3bfdf96f-5856-4d8c-8676-b3ff7d44f0b3","_cell_guid":"c5148c72-aee3-4761-b25d-711e9546430c","collapsed":false,"execution":{"iopub.status.busy":"2023-03-08T12:00:00.674768Z","iopub.execute_input":"2023-03-08T12:00:00.675181Z","iopub.status.idle":"2023-03-08T12:00:00.731364Z","shell.execute_reply.started":"2023-03-08T12:00:00.675151Z","shell.execute_reply":"2023-03-08T12:00:00.730287Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 5: Use the pre-trained model to make predictions on the test set\n\nThe test set is provided as a list of audio file paths. We'll use our `extract_features` function to extract the necessary features from each audio file and then use the pre-trained model to make predictions.","metadata":{"_uuid":"044f1712-b903-471a-a513-bc05c9bbd9bf","_cell_guid":"382994a9-a418-455b-bf00-b80d2de3a413","trusted":true}},{"cell_type":"code","source":"# Load test file paths\ntest_files = glob.glob(\"/kaggle/input/birdclef-2023/test_audio/*.ogg\")\n\n# Generate features and make predictions for each file\npredictions = []\nfor file_path in test_files:\n    # Extract features\n    features = extract_features(file_path)\n    \n    # Reshape features for input to model\n    features = np.expand_dims(features, axis=-1)\n    features = np.expand_dims(features, axis=0)\n    \n    # Make prediction\n    prediction = model.predict(features)\n    predictions.append(prediction)\n\n# Convert predictions to labels\nwith open(labels_path, \"r\") as f:\n    reader = csv.reader(f)\n    next(reader)\n    labels = [row[1] for row in reader]\n\nlabel_predictions = []\nfor prediction in predictions:\n    # Get the index of the class with the highest score\n    max_index = np.argmax(prediction)\n    \n    # Get the name of the predicted class\n    predicted_label = labels[max_index]\n    label_predictions.append(predicted_label)\n\nprint(label_predictions)","metadata":{"_uuid":"3980347e-e60a-41aa-8579-2843d3d1bfce","_cell_guid":"c8a073a3-5aca-40ef-811f-b8d7fa0f1aca","collapsed":false,"execution":{"iopub.status.busy":"2023-03-08T12:00:00.732797Z","iopub.execute_input":"2023-03-08T12:00:00.733376Z","iopub.status.idle":"2023-03-08T12:00:00.752802Z","shell.execute_reply.started":"2023-03-08T12:00:00.733342Z","shell.execute_reply":"2023-03-08T12:00:00.751774Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}