{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":44224,"databundleVersionId":5188730,"sourceType":"competition"},{"sourceId":5467599,"sourceType":"datasetVersion","datasetId":3158155},{"sourceId":3836,"sourceType":"modelInstanceVersion","modelInstanceId":2739,"modelId":319}],"dockerImageVersionId":30407,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook, we'll use a pre-trained machine learning model to generate a submission to the [BirdClef2023 competition](https://www.kaggle.com/c/birdclef-2023).  The goal of the competition is to identify Eastern African bird species by sound.\n\nForked from [Phil Culton's notebook](https://www.kaggle.com/code/philculliton/inferring-birds-with-kaggle-models/notebook)","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\n# import librosa\nfrom IPython.display import Audio\nimport tensorflow_hub as hub\nimport tensorflow_io as tfio\n\nimport pandas as pd\nimport numpy as np\nimport librosa\nimport glob\n\nimport csv\nimport io\nimport os\n\nfrom IPython.display import Audio\nimport matplotlib.pyplot as plt\nfrom PIL import Image","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-11-22T12:59:59.358171Z","iopub.execute_input":"2024-11-22T12:59:59.358958Z","iopub.status.idle":"2024-11-22T12:59:59.366534Z","shell.execute_reply.started":"2024-11-22T12:59:59.358913Z","shell.execute_reply":"2024-11-22T12:59:59.365110Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#spectrogram helper variables and functions\n\n#STFT\nsr=32000\n#window_size = 5.0/1000 #time in seconds\nwindow_length = 2048#int(window_size * sr) #number of samples\noverlap = 0.75\nhop_length = int(window_length*(1-overlap))#hop length\n\nn_fft = int(2**np.ceil(np.log2(window_length)))  #why are we padding with zeros?\ndef get_stft(audio):\n    return np.abs(librosa.stft(y=audio,hop_length=hop_length,win_length=window_length,n_fft=n_fft))\n#norm_STFT = STFT/ (STFT.max()+1e-8)\n#STFT_shape = STFT.shape","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:10.685564Z","iopub.execute_input":"2024-11-22T13:00:10.685983Z","iopub.status.idle":"2024-11-22T13:00:10.693894Z","shell.execute_reply.started":"2024-11-22T13:00:10.685946Z","shell.execute_reply":"2024-11-22T13:00:10.692243Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The competition includes 264 classes of birds, 261 of which exist in this model. We'll set up a way to map the model's output logits to our competition.","metadata":{"execution":{"iopub.status.busy":"2023-04-20T11:23:12.997964Z","iopub.execute_input":"2023-04-20T11:23:12.999300Z","iopub.status.idle":"2023-04-20T11:23:13.007505Z","shell.execute_reply.started":"2023-04-20T11:23:12.999247Z","shell.execute_reply":"2023-04-20T11:23:13.005568Z"}}},{"cell_type":"code","source":"model = hub.load('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1')\nlabels_path = hub.resolve('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1') + \"/assets/label.csv\"\n# Find the name of the class with the top score when mean-aggregated across frames.\ndef class_names_from_csv(class_map_csv_text):\n    \"\"\"Returns list of class names corresponding to score vector.\"\"\"\n    with open(labels_path) as csv_file:\n        csv_reader = csv.reader(csv_file, delimiter=',')\n        class_names = [mid for mid, desc in csv_reader]\n        return class_names[1:]\n\n## note that the bird classifier classifies a much larger set of birds than the\n## competition, so we need to load the model's set of class names or else our \n## indices will be off.\nclasses = class_names_from_csv(labels_path)\n\ntrain_metadata = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\ntrain_metadata.head()\ncompetition_classes = sorted(train_metadata.primary_label.unique())\n\nforced_defaults = 0\ncompetition_class_map = []\nfor c in competition_classes:\n    try:\n        i = classes.index(c)\n        competition_class_map.append(i)\n    except:\n        competition_class_map.append(0)\n        forced_defaults += 1\n        \n## this is the count of classes not supported by our pretrained model\n## you could choose to simply not predict these, set a default as above,\n## or create your own model using the pretrained model as a base.\nforced_defaults\n\ndef frame_audio(\n      audio_array: np.ndarray,\n      window_size_s: float = 5.0,\n      hop_size_s: float = 5.0,\n      sample_rate = 32000,\n      ) -> np.ndarray:\n    \n    \"\"\"Helper function for framing audio for inference.\"\"\"\n    \"\"\" using tf.signal \"\"\"\n    if window_size_s is None or window_size_s < 0:\n        return audio_array[np.newaxis, :]\n    frame_length = int(window_size_s * sample_rate)\n    hop_length = int(hop_size_s * sample_rate)\n    framed_audio = tf.signal.frame(audio_array, frame_length, hop_length, pad_end=True)\n    return framed_audio\n\ndef ensure_sample_rate(waveform, original_sample_rate,\n                       desired_sample_rate=32000):\n    \"\"\"Resample waveform if required.\"\"\"\n    if original_sample_rate != desired_sample_rate:\n        waveform = tfio.audio.resample(waveform, original_sample_rate, desired_sample_rate)\n    return desired_sample_rate, waveform","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:11.840135Z","iopub.execute_input":"2024-11-22T13:00:11.841571Z","iopub.status.idle":"2024-11-22T13:00:21.363217Z","shell.execute_reply.started":"2024-11-22T13:00:11.841514Z","shell.execute_reply":"2024-11-22T13:00:21.361914Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# os.listdir('/kaggle/input/birdclef-2023/train_audio/hartur1/XC537900.ogg')[0]","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:22.003713Z","iopub.execute_input":"2024-11-22T13:00:22.004149Z","iopub.status.idle":"2024-11-22T13:00:22.009844Z","shell.execute_reply.started":"2024-11-22T13:00:22.004109Z","shell.execute_reply":"2024-11-22T13:00:22.008472Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"\n\nWe'll start by loading one of five of examples and using the IPython.display.Audio module to play them, show the waveform and spectrogram then predict!","metadata":{}},{"cell_type":"code","source":"'''\n0 - Common Bulbul - combul2\n1 - Tropical Boubou - trobou1\n2 - Marabou Stork - marsto1   #Misclassified\n3 - Pied Crow - piecro1 #Misclassified\n4 - Hartlaub's Turaco - hartur1 \n'''\naudio_list = [\"/kaggle/input/birdclef-2023/train_audio/combul2/XC115401.ogg\",\"/kaggle/input/birdclef-2023/train_audio/trobou1/XC608026.ogg\",\n             \"/kaggle/input/birdclef-2023/train_audio/marsto1/XC508624.ogg\",\"/kaggle/input/birdclef-2023/train_audio/piecro1/XC126577.ogg\",\n             \"/kaggle/input/birdclef-2023/train_audio/hartur1/XC122257.ogg\"]\n\n'''\n#Misclassified\n\n/kaggle/input/birdclef-2023/train_audio/hartur1/XC122256.ogg\n/kaggle/input/birdclef-2023/train_audio/piecro1/XC117040.ogg\n\n\n'''","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:23.293330Z","iopub.execute_input":"2024-11-22T13:00:23.294874Z","iopub.status.idle":"2024-11-22T13:00:23.305169Z","shell.execute_reply.started":"2024-11-22T13:00:23.294797Z","shell.execute_reply":"2024-11-22T13:00:23.303684Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load a sample audio files from two different species\naudio_ex, sr_ex = librosa.load(audio_list[0],duration=10.0)\n#audio_abh, sr_abh = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/abhori1/XC127317.ogg\")","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:24.167815Z","iopub.execute_input":"2024-11-22T13:00:24.169217Z","iopub.status.idle":"2024-11-22T13:00:37.880090Z","shell.execute_reply.started":"2024-11-22T13:00:24.169164Z","shell.execute_reply":"2024-11-22T13:00:37.878323Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sr_ex","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:37.883074Z","iopub.execute_input":"2024-11-22T13:00:37.883911Z","iopub.status.idle":"2024-11-22T13:00:37.893110Z","shell.execute_reply.started":"2024-11-22T13:00:37.883865Z","shell.execute_reply":"2024-11-22T13:00:37.891559Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Play the audio \nAudio(data=audio_ex, rate=sr_ex)","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:37.894955Z","iopub.execute_input":"2024-11-22T13:00:37.895492Z","iopub.status.idle":"2024-11-22T13:00:37.938581Z","shell.execute_reply.started":"2024-11-22T13:00:37.895423Z","shell.execute_reply":"2024-11-22T13:00:37.936627Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#waveform\nlibrosa.display.waveshow(audio_ex)","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:37.943085Z","iopub.execute_input":"2024-11-22T13:00:37.943704Z","iopub.status.idle":"2024-11-22T13:00:38.472136Z","shell.execute_reply.started":"2024-11-22T13:00:37.943649Z","shell.execute_reply":"2024-11-22T13:00:38.470919Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#spectrogram\nSTFT = get_stft(audio_ex)\naudio_db = librosa.amplitude_to_db(STFT)\nlibrosa.display.specshow(data=audio_db,y_axis=\"mel\",x_axis=\"time\", sr=sr,win_length=window_length,hop_length=hop_length)","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:38.473648Z","iopub.execute_input":"2024-11-22T13:00:38.474003Z","iopub.status.idle":"2024-11-22T13:00:39.336920Z","shell.execute_reply.started":"2024-11-22T13:00:38.473969Z","shell.execute_reply":"2024-11-22T13:00:39.335566Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Below we load one training sample - use the Audio function to listen to the samples inside the notebook!","metadata":{}},{"cell_type":"code","source":"sample_rate, wav_data = ensure_sample_rate(audio_ex, sr_ex)\nAudio(wav_data, rate=sample_rate)","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:39.338689Z","iopub.execute_input":"2024-11-22T13:00:39.339105Z","iopub.status.idle":"2024-11-22T13:00:39.961589Z","shell.execute_reply.started":"2024-11-22T13:00:39.339066Z","shell.execute_reply":"2024-11-22T13:00:39.959956Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Make predictions\n\n We use the pretrained model to return probabilities for all 10k birds included in the model, then pull out the classes used in this competition to create a final submission row. Note that we are NOT doing anything special to handle the 3 missing classes; those will need fine-tuning / transfer learning, which will be handled in a separate notebook.","metadata":{}},{"cell_type":"code","source":"fixed_tm = frame_audio(wav_data)\nlogits, embeddings = model.infer_tf(fixed_tm[:1])\n\nprobabilities = tf.nn.softmax(logits)\nargmax = np.argmax(probabilities)\n\nprint(f\"The audio is from the class {classes[argmax]} with probability of {probabilities[0][argmax]}\")\n\nimg = np.asarray(Image.open(\"/kaggle/input/tech-tutors-presentation-images/\"+str(classes[argmax])+\".jpg\"))\nplt.imshow(img);","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:39.964597Z","iopub.execute_input":"2024-11-22T13:00:39.965104Z","iopub.status.idle":"2024-11-22T13:00:52.592525Z","shell.execute_reply.started":"2024-11-22T13:00:39.965057Z","shell.execute_reply":"2024-11-22T13:00:52.590996Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# THE END","metadata":{}},{"cell_type":"code","source":"def predict_for_sample(filename, sample_submission, frame_limit_secs=None):\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    \n    audio, sample_rate = librosa.load(filename)\n    sample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\n    \n    fixed_tm = frame_audio(wav_data)\n    \n    frame = 5\n    all_logits, all_embeddings = model.infer_tf(fixed_tm[:1])\n    for window in fixed_tm[1:]:\n        if frame_limit_secs and frame > frame_limit_secs:\n            continue\n        \n        logits, embeddings = model.infer_tf(window[np.newaxis, :])\n        all_logits = np.concatenate([all_logits, logits], axis=0)\n        frame += 5\n    \n    frame = 5\n    all_probabilities = []\n    for frame_logits in all_logits:\n        probabilities = tf.nn.softmax(frame_logits).numpy()\n        \n        ## set the appropriate row in the sample submission\n        sample_submission.loc[sample_submission.row_id == file_id + \"_\" + str(frame), competition_classes] = probabilities[competition_class_map]\n        frame += 5","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:52.595203Z","iopub.execute_input":"2024-11-22T13:00:52.595839Z","iopub.status.idle":"2024-11-22T13:00:52.612522Z","shell.execute_reply.started":"2024-11-22T13:00:52.595771Z","shell.execute_reply":"2024-11-22T13:00:52.610147Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Step 6: Generate a submission\n\nNow we process all of the test samples as discussed above, creating output rows, and saving them in the provided `sample_submission.csv`. Finally, we save these rows to our final output file: `submission.csv`. This is the file that gets submitted and scored when you submit the notebook.","metadata":{}},{"cell_type":"code","source":"test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\ntest_samples","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:52.614663Z","iopub.execute_input":"2024-11-22T13:00:52.615124Z","iopub.status.idle":"2024-11-22T13:00:52.650083Z","shell.execute_reply.started":"2024-11-22T13:00:52.615087Z","shell.execute_reply":"2024-11-22T13:00:52.648663Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\nsample_sub[competition_classes] = sample_sub[competition_classes].astype(np.float32)\nsample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:52.652797Z","iopub.execute_input":"2024-11-22T13:00:52.653197Z","iopub.status.idle":"2024-11-22T13:00:52.763991Z","shell.execute_reply.started":"2024-11-22T13:00:52.653160Z","shell.execute_reply":"2024-11-22T13:00:52.762190Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"frame_limit_secs = 15 if sample_sub.shape[0] == 3 else None\nfor sample_filename in test_samples:\n    predict_for_sample(sample_filename, sample_sub, frame_limit_secs=15)","metadata":{"execution":{"iopub.status.busy":"2024-11-22T13:00:52.765622Z","iopub.execute_input":"2024-11-22T13:00:52.766014Z","iopub.status.idle":"2024-11-22T13:01:06.219602Z","shell.execute_reply.started":"2024-11-22T13:00:52.765979Z","shell.execute_reply":"2024-11-22T13:01:06.218323Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T13:01:06.221122Z","iopub.execute_input":"2024-11-22T13:01:06.221509Z","iopub.status.idle":"2024-11-22T13:01:06.254810Z","shell.execute_reply.started":"2024-11-22T13:01:06.221434Z","shell.execute_reply":"2024-11-22T13:01:06.253471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T13:01:06.256613Z","iopub.execute_input":"2024-11-22T13:01:06.256991Z","iopub.status.idle":"2024-11-22T13:01:06.282199Z","shell.execute_reply.started":"2024-11-22T13:01:06.256956Z","shell.execute_reply":"2024-11-22T13:01:06.280755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip freeze > requirements.txt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-22T13:01:06.284200Z","iopub.execute_input":"2024-11-22T13:01:06.284752Z","iopub.status.idle":"2024-11-22T13:01:10.588996Z","shell.execute_reply.started":"2024-11-22T13:01:06.284695Z","shell.execute_reply":"2024-11-22T13:01:10.586293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}