{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":70203,"databundleVersionId":8068726},{"sourceType":"modelInstanceVersion","sourceId":3836,"databundleVersionId":5146208,"modelInstanceId":2739},{"sourceType":"modelInstanceVersion","sourceId":15853,"databundleVersionId":7931683,"modelInstanceId":2739}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-04-14T09:06:06.135004Z","iopub.execute_input":"2024-04-14T09:06:06.135464Z","iopub.status.idle":"2024-04-14T09:06:08.464683Z","shell.execute_reply.started":"2024-04-14T09:06:06.135435Z","shell.execute_reply":"2024-04-14T09:06:08.463143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_io as tfio\n\nimport pandas as pd\nimport numpy as np\nimport librosa\nimport glob\n\nimport csv\nimport io\n\nfrom IPython.display import Audio","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:06:08.467085Z","iopub.execute_input":"2024-04-14T09:06:08.467939Z","iopub.status.idle":"2024-04-14T09:06:08.473886Z","shell.execute_reply.started":"2024-04-14T09:06:08.467895Z","shell.execute_reply":"2024-04-14T09:06:08.472923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_abe, sr_abe = librosa.load(\"/kaggle/input/birdclef-2024/train_audio/asbfly/XC134896.ogg\")\naudio_abh, sr_abh = librosa.load(\"/kaggle/input/birdclef-2024/train_audio/ashdro1/XC114598.ogg\")","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:06:08.475009Z","iopub.execute_input":"2024-04-14T09:06:08.476110Z","iopub.status.idle":"2024-04-14T09:06:08.661495Z","shell.execute_reply.started":"2024-04-14T09:06:08.476070Z","shell.execute_reply":"2024-04-14T09:06:08.659898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abe, rate=sr_abe)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:06:08.664110Z","iopub.execute_input":"2024-04-14T09:06:08.665008Z","iopub.status.idle":"2024-04-14T09:06:08.699867Z","shell.execute_reply.started":"2024-04-14T09:06:08.664972Z","shell.execute_reply":"2024-04-14T09:06:08.698288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abh, rate=sr_abh)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:06:08.701764Z","iopub.execute_input":"2024-04-14T09:06:08.702546Z","iopub.status.idle":"2024-04-14T09:06:08.773553Z","shell.execute_reply.started":"2024-04-14T09:06:08.702505Z","shell.execute_reply":"2024-04-14T09:06:08.772528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = hub.load('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1')\nlabels_path = hub.resolve('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1') + \"/assets/label.csv\"","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:06:08.774631Z","iopub.execute_input":"2024-04-14T09:06:08.775650Z","iopub.status.idle":"2024-04-14T09:06:26.629935Z","shell.execute_reply.started":"2024-04-14T09:06:08.775613Z","shell.execute_reply":"2024-04-14T09:06:26.628760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find the name of the class with the top score when mean-aggregated across frames.\ndef class_names_from_csv(class_map_csv_text):\n    \"\"\"Returns list of class names corresponding to score vector.\"\"\"\n    with open(labels_path) as csv_file:\n        csv_reader = csv.reader(csv_file, delimiter=',')\n        class_names = [mid for mid, desc in csv_reader]\n        return class_names[1:]\n\n## note that the bird classifier classifies a much larger set of birds than the\n## competition, so we need to load the model's set of class names or else our \n## indices will be off.\nclasses = class_names_from_csv(labels_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:12:21.699158Z","iopub.execute_input":"2024-04-14T09:12:21.699770Z","iopub.status.idle":"2024-04-14T09:12:21.726249Z","shell.execute_reply.started":"2024-04-14T09:12:21.699725Z","shell.execute_reply":"2024-04-14T09:12:21.724405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata = pd.read_csv(\"/kaggle/input/birdclef-2024/train_metadata.csv\")\ntrain_metadata.head()\ncompetition_classes = sorted(train_metadata.primary_label.unique())\n\nforced_defaults = 0\ncompetition_class_map = []\nfor c in competition_classes:\n    try:\n        i = classes.index(c)\n        competition_class_map.append(i)\n    except:\n        competition_class_map.append(0)\n        forced_defaults += 1\n        \n## this is the count of classes not supported by our pretrained model\n## you could choose to simply not predict these, set a default as above,\n## or create your own model using the pretrained model as a base.\nforced_defaults","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:12:49.614033Z","iopub.execute_input":"2024-04-14T09:12:49.614476Z","iopub.status.idle":"2024-04-14T09:12:49.875935Z","shell.execute_reply.started":"2024-04-14T09:12:49.614446Z","shell.execute_reply":"2024-04-14T09:12:49.874597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def frame_audio(\n      audio_array: np.ndarray,\n      window_size_s: float = 5.0,\n      hop_size_s: float = 5.0,\n      sample_rate = 32000,\n      ) -> np.ndarray:\n    \n    \"\"\"Helper function for framing audio for inference.\"\"\"\n    \"\"\" using tf.signal \"\"\"\n    if window_size_s is None or window_size_s < 0:\n        return audio_array[np.newaxis, :]\n    frame_length = int(window_size_s * sample_rate)\n    hop_length = int(hop_size_s * sample_rate)\n    framed_audio = tf.signal.frame(audio_array, frame_length, hop_length, pad_end=True)\n    return framed_audio\n\ndef ensure_sample_rate(waveform, original_sample_rate,\n                       desired_sample_rate=32000):\n    \"\"\"Resample waveform if required.\"\"\"\n    if original_sample_rate != desired_sample_rate:\n        waveform = tfio.audio.resample(waveform, original_sample_rate, desired_sample_rate)\n    return desired_sample_rate, waveform","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:13:02.181560Z","iopub.execute_input":"2024-04-14T09:13:02.182095Z","iopub.status.idle":"2024-04-14T09:13:02.192277Z","shell.execute_reply.started":"2024-04-14T09:13:02.182058Z","shell.execute_reply":"2024-04-14T09:13:02.190798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio, sample_rate = librosa.load(\"/kaggle/input/birdclef-2024/train_audio/ashpri1/XC116338.ogg\")\nsample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\nAudio(wav_data, rate=sample_rate)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:13:40.445937Z","iopub.execute_input":"2024-04-14T09:13:40.446423Z","iopub.status.idle":"2024-04-14T09:13:41.814224Z","shell.execute_reply.started":"2024-04-14T09:13:40.446389Z","shell.execute_reply":"2024-04-14T09:13:41.813243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fixed_tm = frame_audio(wav_data)\nlogits, embeddings = model.infer_tf(fixed_tm[:1])\nprobabilities = tf.nn.softmax(logits)\nargmax = np.argmax(probabilities)\nprint(f\"The audio is from the class {classes[argmax]} (element:{argmax} in the label.csv file), with probability of {probabilities[0][argmax]}\")","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:14:00.119711Z","iopub.execute_input":"2024-04-14T09:14:00.120136Z","iopub.status.idle":"2024-04-14T09:14:09.071819Z","shell.execute_reply.started":"2024-04-14T09:14:00.120103Z","shell.execute_reply":"2024-04-14T09:14:09.070871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_for_sample(filename, sample_submission, frame_limit_secs=None):\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    \n    audio, sample_rate = librosa.load(filename)\n    sample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\n    \n    fixed_tm = frame_audio(wav_data)\n    \n    frame = 5\n    all_logits, all_embeddings = model.infer_tf(fixed_tm[:1])\n    for window in fixed_tm[1:]:\n        if frame_limit_secs and frame > frame_limit_secs:\n            continue\n        \n        logits, embeddings = model.infer_tf(window[np.newaxis, :])\n        all_logits = np.concatenate([all_logits, logits], axis=0)\n        frame += 5\n    \n    frame = 5\n    all_probabilities = []\n    for frame_logits in all_logits:\n        probabilities = tf.nn.softmax(frame_logits).numpy()\n        \n        ## set the appropriate row in the sample submission\n        sample_submission.loc[sample_submission.row_id == file_id + \"_\" + str(frame), competition_classes] = probabilities[competition_class_map]\n        frame += 5","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:14:47.459936Z","iopub.execute_input":"2024-04-14T09:14:47.460389Z","iopub.status.idle":"2024-04-14T09:14:47.471810Z","shell.execute_reply.started":"2024-04-14T09:14:47.460345Z","shell.execute_reply":"2024-04-14T09:14:47.470460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_samples = list(glob.glob(\"/kaggle/input/birdclef-2024/test_soundscapes/*.ogg\"))\ntest_samples\n","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:15:34.383647Z","iopub.execute_input":"2024-04-14T09:15:34.384080Z","iopub.status.idle":"2024-04-14T09:15:34.394033Z","shell.execute_reply.started":"2024-04-14T09:15:34.384041Z","shell.execute_reply":"2024-04-14T09:15:34.392708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\")\nsample_sub[competition_classes] = sample_sub[competition_classes].astype(np.float32)\nsample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:16:12.999086Z","iopub.execute_input":"2024-04-14T09:16:13.000672Z","iopub.status.idle":"2024-04-14T09:16:13.089927Z","shell.execute_reply.started":"2024-04-14T09:16:13.000627Z","shell.execute_reply":"2024-04-14T09:16:13.088262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frame_limit_secs = 15 if sample_sub.shape[0] == 3 else None\nfor sample_filename in test_samples:\n    predict_for_sample(sample_filename, sample_sub, frame_limit_secs=15)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:16:28.246401Z","iopub.execute_input":"2024-04-14T09:16:28.246883Z","iopub.status.idle":"2024-04-14T09:16:28.254259Z","shell.execute_reply.started":"2024-04-14T09:16:28.246852Z","shell.execute_reply":"2024-04-14T09:16:28.252171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:16:37.900219Z","iopub.execute_input":"2024-04-14T09:16:37.901509Z","iopub.status.idle":"2024-04-14T09:16:37.933464Z","shell.execute_reply.started":"2024-04-14T09:16:37.901472Z","shell.execute_reply":"2024-04-14T09:16:37.932125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-04-14T09:16:46.251528Z","iopub.execute_input":"2024-04-14T09:16:46.252582Z","iopub.status.idle":"2024-04-14T09:16:46.272077Z","shell.execute_reply.started":"2024-04-14T09:16:46.252541Z","shell.execute_reply":"2024-04-14T09:16:46.270994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}