{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-07T23:54:46.468184Z","iopub.execute_input":"2023-05-07T23:54:46.468719Z","iopub.status.idle":"2023-05-07T23:54:46.813424Z","shell.execute_reply.started":"2023-05-07T23:54:46.468687Z","shell.execute_reply":"2023-05-07T23:54:46.812294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Import required libraries","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_io as tfio\n\nimport pandas as pd\nimport numpy as np\nimport librosa\nimport glob\n\nimport csv\nimport io\n\nfrom IPython.display import Audio","metadata":{"execution":{"iopub.status.busy":"2023-05-07T23:54:46.814643Z","iopub.execute_input":"2023-05-07T23:54:46.814973Z","iopub.status.idle":"2023-05-07T23:54:49.922124Z","shell.execute_reply.started":"2023-05-07T23:54:46.814945Z","shell.execute_reply":"2023-05-07T23:54:49.920907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load a sample audio files from two different species\naudio_abe, sr_abe = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/abethr1/XC128013.ogg\")\naudio_abh, sr_abh = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/abhori1/XC127317.ogg\")\naudio_sun, sr_sun = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/edcsun3/XC470591.ogg\")","metadata":{"execution":{"iopub.status.busy":"2023-05-07T23:54:49.923556Z","iopub.execute_input":"2023-05-07T23:54:49.924236Z","iopub.status.idle":"2023-05-07T23:54:52.292732Z","shell.execute_reply.started":"2023-05-07T23:54:49.924203Z","shell.execute_reply":"2023-05-07T23:54:52.291564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abe, rate=sr_abe)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T23:54:52.294993Z","iopub.execute_input":"2023-05-07T23:54:52.295654Z","iopub.status.idle":"2023-05-07T23:54:52.353259Z","shell.execute_reply.started":"2023-05-07T23:54:52.295620Z","shell.execute_reply":"2023-05-07T23:54:52.352316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abh, rate=sr_abh)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T23:54:52.354369Z","iopub.execute_input":"2023-05-07T23:54:52.355146Z","iopub.status.idle":"2023-05-07T23:54:52.415808Z","shell.execute_reply.started":"2023-05-07T23:54:52.355107Z","shell.execute_reply":"2023-05-07T23:54:52.413682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_sun, rate=sr_sun)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T23:54:52.417409Z","iopub.execute_input":"2023-05-07T23:54:52.417808Z","iopub.status.idle":"2023-05-07T23:54:52.451033Z","shell.execute_reply.started":"2023-05-07T23:54:52.417774Z","shell.execute_reply":"2023-05-07T23:54:52.449884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load the model.\nmodel = hub.load('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1')\n# model = hub.load('https://tfhub.dev/google/yamnet/1')\nlabels_path = hub.resolve('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1') + \"/assets/label.csv\"","metadata":{"execution":{"iopub.status.busy":"2023-05-07T23:54:52.452494Z","iopub.execute_input":"2023-05-07T23:54:52.452899Z","iopub.status.idle":"2023-05-07T23:54:58.959813Z","shell.execute_reply.started":"2023-05-07T23:54:52.452859Z","shell.execute_reply":"2023-05-07T23:54:58.958411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a function to extract class names from the provided CSV file.\ndef class_names_from_csv(class_map_csv_text):\n    \"\"\"Returns list of class names corresponding to score vector.\"\"\"\n    with open(labels_path) as csv_file:\n        csv_reader = csv.reader(csv_file, delimiter=',')\n        # Extract the class names from the CSV file.\n        class_names = [mid for mid, desc in csv_reader]\n        # Return the class names starting from index 1 (i.e. skipping the first row which contains the header).\n        return class_names[1:]\n\n# Call the function to extract class names from the provided CSV file.\n## Note that the bird classifier classifies a much larger set of birds than the\n## competition, so we need to load the model's set of class names or else our \n## indices will be off.\nclasses = class_names_from_csv(labels_path)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-07T23:57:01.618661Z","iopub.execute_input":"2023-05-07T23:57:01.619106Z","iopub.status.idle":"2023-05-07T23:57:01.645373Z","shell.execute_reply.started":"2023-05-07T23:57:01.619051Z","shell.execute_reply":"2023-05-07T23:57:01.643836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\ntrain_metadata.head()\n\n# get the unique primary labels from train_metadata\ncompetition_classes = sorted(train_metadata.primary_label.unique())\n\n# map the competition classes to the pretrained model's class indices\n# use a default index of 0 for classes not supported by the pretrained model\ncompetition_class_map = [classes.index(c) if c in classes else 0 for c in competition_classes]\n\n# count the number of classes not supported by the pretrained model\nnum_unsupported_classes = competition_class_map.count(0)\nnum_unsupported_classes","metadata":{"execution":{"iopub.status.busy":"2023-05-07T23:58:21.538417Z","iopub.execute_input":"2023-05-07T23:58:21.538823Z","iopub.status.idle":"2023-05-07T23:58:21.706721Z","shell.execute_reply.started":"2023-05-07T23:58:21.538794Z","shell.execute_reply":"2023-05-07T23:58:21.705461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def frame_audio(audio_array: np.ndarray,\n                window_size_s: float = 5.0,\n                hop_size_s: float = 5.0,\n                sample_rate=32000) -> np.ndarray:\n    \"\"\"Framing audio for inference.\n\n    Args:\n        audio_array (np.ndarray): The audio signal.\n        window_size_s (float, optional): The window size in seconds. Defaults to 5.0.\n        hop_size_s (float, optional): The hop size in seconds. Defaults to 5.0.\n        sample_rate (int, optional): The sample rate of the audio. Defaults to 32000.\n\n    Returns:\n        np.ndarray: The framed audio signal.\n    \"\"\"\n    if window_size_s is None or window_size_s < 0:\n        return audio_array[np.newaxis, :]\n\n    frame_length = int(window_size_s * sample_rate)\n    hop_length = int(hop_size_s * sample_rate)\n\n    # Frame the audio using TensorFlow Signal.\n    framed_audio = tf.signal.frame(audio_array, frame_length, hop_length, pad_end=True)\n    return framed_audio\n\n\ndef ensure_sample_rate(waveform, original_sample_rate, desired_sample_rate=32000):\n    \"\"\"Resample the waveform if required.\n\n    Args:\n        waveform (tf.Tensor): The audio signal.\n        original_sample_rate (int): The original sample rate of the audio.\n        desired_sample_rate (int, optional): The desired sample rate. Defaults to 32000.\n\n    Returns:\n        Tuple[int, tf.Tensor]: A tuple containing the new sample rate and the resampled waveform.\n    \"\"\"\n    if original_sample_rate != desired_sample_rate:\n        waveform = tfio.audio.resample(waveform, original_sample_rate, desired_sample_rate)\n    return desired_sample_rate, waveform\n","metadata":{"execution":{"iopub.status.busy":"2023-05-07T23:59:42.579938Z","iopub.execute_input":"2023-05-07T23:59:42.580332Z","iopub.status.idle":"2023-05-07T23:59:42.588570Z","shell.execute_reply.started":"2023-05-07T23:59:42.580302Z","shell.execute_reply":"2023-05-07T23:59:42.587660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio, sample_rate = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/blcapa2/XC120989.ogg\")\nsample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\nAudio(wav_data, rate=sample_rate)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:06:12.662286Z","iopub.execute_input":"2023-05-08T00:06:12.662806Z","iopub.status.idle":"2023-05-08T00:06:13.121805Z","shell.execute_reply.started":"2023-05-08T00:06:12.662759Z","shell.execute_reply":"2023-05-08T00:06:13.120274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# frame the audio using a window size of 5 seconds and hop size of 5 seconds\nfixed_tm = frame_audio(wav_data, window_size_s=5.0, hop_size_s=5.0)\n\n# pass the framed audio to the model's infer_tf function to get the logits and embeddings\nlogits, embeddings = model.infer_tf(fixed_tm[:1])\n\n# apply softmax to get probabilities and find the index of the highest probability\nprobabilities = tf.nn.softmax(logits)\nargmax = np.argmax(probabilities)\n\n# print the predicted class name and its corresponding probability\nprint(f\"The audio is from the class {classes[argmax]} (element:{argmax} in the label.csv file), with probability of {probabilities[0][argmax]}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:06:30.045446Z","iopub.execute_input":"2023-05-08T00:06:30.045827Z","iopub.status.idle":"2023-05-08T00:06:32.431502Z","shell.execute_reply.started":"2023-05-08T00:06:30.045799Z","shell.execute_reply":"2023-05-08T00:06:32.430162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_for_sample(filename, sample_submission, frame_limit_secs=None):\n    # Get the file ID from the filename\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    \n    # Load the audio data and ensure the correct sample rate\n    audio, sample_rate = librosa.load(filename)\n    sample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\n    \n    # Frame the audio data into fixed-size windows\n    fixed_tm = frame_audio(wav_data)\n    \n    # Predict the class probabilities for each window\n    frame = 5\n    all_logits, all_embeddings = model.infer_tf(fixed_tm[:1])\n    for window in fixed_tm[1:]:\n        if frame_limit_secs and frame > frame_limit_secs:\n            continue\n        \n        # Infer the logits and embeddings for the current window\n        logits, embeddings = model.infer_tf(window[np.newaxis, :])\n        \n        # Append the logits to the array of all logits\n        all_logits = np.concatenate([all_logits, logits], axis=0)\n        \n        # Increment the frame counter\n        frame += 5\n    \n    # Convert the logits to probabilities and update the sample submission\n    frame = 5\n    all_probabilities = []\n    for frame_logits in all_logits:\n        probabilities = tf.nn.softmax(frame_logits).numpy()\n        \n        ## set the appropriate row in the sample submission\n        sample_submission.loc[sample_submission.row_id == file_id + \"_\" + str(frame), competition_classes] = probabilities[competition_class_map]\n        \n        # Increment the frame counter\n        frame += 5\n","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:08:26.842909Z","iopub.execute_input":"2023-05-08T00:08:26.843953Z","iopub.status.idle":"2023-05-08T00:08:26.852515Z","shell.execute_reply.started":"2023-05-08T00:08:26.843916Z","shell.execute_reply":"2023-05-08T00:08:26.851607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\ntest_samples","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:08:46.782438Z","iopub.execute_input":"2023-05-08T00:08:46.782857Z","iopub.status.idle":"2023-05-08T00:08:46.790609Z","shell.execute_reply.started":"2023-05-08T00:08:46.782824Z","shell.execute_reply":"2023-05-08T00:08:46.789815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\nsample_sub[competition_classes] = sample_sub[competition_classes].astype(np.float32)\nsample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:12:44.245430Z","iopub.execute_input":"2023-05-08T00:12:44.245808Z","iopub.status.idle":"2023-05-08T00:12:44.357582Z","shell.execute_reply.started":"2023-05-08T00:12:44.245781Z","shell.execute_reply":"2023-05-08T00:12:44.356535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frame_limit_secs = 15 if sample_sub.shape[0] == 3 else None\nfor sample_filename in test_samples:\n    predict_for_sample(sample_filename, sample_sub, frame_limit_secs=15)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:14:16.798135Z","iopub.execute_input":"2023-05-08T00:14:16.798534Z","iopub.status.idle":"2023-05-08T00:14:29.638712Z","shell.execute_reply.started":"2023-05-08T00:14:16.798506Z","shell.execute_reply":"2023-05-08T00:14:29.637400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:14:34.342462Z","iopub.execute_input":"2023-05-08T00:14:34.342892Z","iopub.status.idle":"2023-05-08T00:14:34.379557Z","shell.execute_reply.started":"2023-05-08T00:14:34.342855Z","shell.execute_reply":"2023-05-08T00:14:34.378305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:15:08.146228Z","iopub.execute_input":"2023-05-08T00:15:08.146742Z","iopub.status.idle":"2023-05-08T00:15:08.167926Z","shell.execute_reply.started":"2023-05-08T00:15:08.146699Z","shell.execute_reply":"2023-05-08T00:15:08.166943Z"},"trusted":true},"execution_count":null,"outputs":[]}]}