{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In this notebook, I will walk through the dataset and generate a baseline model for [BirdClef2023 competition](https://www.kaggle.com/c/birdclef-2023). The goal of the competition is to identify Eastern African bird species by sound.","metadata":{}},{"cell_type":"markdown","source":"## Step 1: Imports","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_io as tfio\nimport geopandas as gpd\nfrom geopandas import GeoDataFrame\nfrom shapely.geometry import Point\nimport pandas as pd\nimport numpy as np\nimport librosa\nimport glob\n\nimport csv\nimport io\n\nfrom IPython.display import Audio","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-15T11:45:48.664710Z","iopub.execute_input":"2023-03-15T11:45:48.665097Z","iopub.status.idle":"2023-03-15T11:45:59.066183Z","shell.execute_reply.started":"2023-03-15T11:45:48.665054Z","shell.execute_reply":"2023-03-15T11:45:59.064893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/birdclef-2023/train_metadata.csv')\nprint(f\"train_shape:{train_df.shape}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-15T11:45:59.067762Z","iopub.execute_input":"2023-03-15T11:45:59.068426Z","iopub.status.idle":"2023-03-15T11:45:59.191034Z","shell.execute_reply.started":"2023-03-15T11:45:59.068392Z","shell.execute_reply":"2023-03-15T11:45:59.190282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Explore how the datatypes look like for all the columns\ncount=0\nfor feature in list(train_df.columns):\n    \n    if train_df[feature].dtype == 'float64':\n        count +=1\n    else:\n        print(feature,train_df[feature].dtype)\n        ","metadata":{"execution":{"iopub.status.busy":"2023-03-15T11:45:59.192210Z","iopub.execute_input":"2023-03-15T11:45:59.193067Z","iopub.status.idle":"2023-03-15T11:45:59.207237Z","shell.execute_reply.started":"2023-03-15T11:45:59.193037Z","shell.execute_reply":"2023-03-15T11:45:59.205590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check out for missing values in training data\nall_features_nan =  train_df.columns[train_df.isna().any()].tolist()\n\n\nfor feature in all_features_nan:\n    print(feature,train_df[feature].isna().sum())","metadata":{"execution":{"iopub.status.busy":"2023-03-15T11:45:59.209819Z","iopub.execute_input":"2023-03-15T11:45:59.210666Z","iopub.status.idle":"2023-03-15T11:45:59.224401Z","shell.execute_reply.started":"2023-03-15T11:45:59.210634Z","shell.execute_reply":"2023-03-15T11:45:59.223311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Group places together by latitude and longtitude\ntrain_df.groupby(['latitude','longitude']).size()","metadata":{"execution":{"iopub.status.busy":"2023-03-15T11:45:59.227035Z","iopub.execute_input":"2023-03-15T11:45:59.227359Z","iopub.status.idle":"2023-03-15T11:45:59.254644Z","shell.execute_reply.started":"2023-03-15T11:45:59.227334Z","shell.execute_reply":"2023-03-15T11:45:59.253854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# coverting csv to geopandas dataframe\ngeometry = [Point(xy) for xy in zip(train_df['longitude'], train_df['latitude'])]\ngdf = GeoDataFrame(train_df, geometry = geometry)","metadata":{"execution":{"iopub.status.busy":"2023-03-15T11:45:59.255564Z","iopub.execute_input":"2023-03-15T11:45:59.256575Z","iopub.status.idle":"2023-03-15T11:45:59.806480Z","shell.execute_reply.started":"2023-03-15T11:45:59.256546Z","shell.execute_reply":"2023-03-15T11:45:59.805500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting naturalearth_lowres map\nworld = gpd.read_file(gpd.datasets.get_path('naturalearth_lowres'))\ngdf.plot(ax = world.plot(figsize = (10, 10)), color = 'brown', markersize = 5)","metadata":{"execution":{"iopub.status.busy":"2023-03-15T11:45:59.807587Z","iopub.execute_input":"2023-03-15T11:45:59.807855Z","iopub.status.idle":"2023-03-15T11:46:02.576107Z","shell.execute_reply.started":"2023-03-15T11:45:59.807830Z","shell.execute_reply":"2023-03-15T11:46:02.575378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Bird Species","metadata":{}},{"cell_type":"code","source":"classes = sorted(train_df.primary_label.unique())\nlen(classes)","metadata":{"execution":{"iopub.status.busy":"2023-03-15T11:46:02.577067Z","iopub.execute_input":"2023-03-15T11:46:02.577913Z","iopub.status.idle":"2023-03-15T11:46:02.585687Z","shell.execute_reply.started":"2023-03-15T11:46:02.577881Z","shell.execute_reply":"2023-03-15T11:46:02.584891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 2: Explore the training data\n\nWe'll start by loading a couple of training examples and using the IPython.display.Audio module to play them!","metadata":{}},{"cell_type":"code","source":"# Load a sample audio files from two different species\naudio_abe, sr_abe = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/abethr1/XC128013.ogg\")\naudio_abh, sr_abh = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/abhori1/XC127317.ogg\")","metadata":{"execution":{"iopub.status.busy":"2023-03-15T11:46:33.321916Z","iopub.execute_input":"2023-03-15T11:46:33.323591Z","iopub.status.idle":"2023-03-15T11:46:33.427895Z","shell.execute_reply.started":"2023-03-15T11:46:33.323545Z","shell.execute_reply":"2023-03-15T11:46:33.426871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abe, rate=sr_abe)","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:24.296369Z","iopub.execute_input":"2023-03-13T06:42:24.297112Z","iopub.status.idle":"2023-03-13T06:42:24.374573Z","shell.execute_reply.started":"2023-03-13T06:42:24.297073Z","shell.execute_reply":"2023-03-13T06:42:24.373515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Play the audio\nAudio(data=audio_abh, rate=sr_abh)","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:24.377155Z","iopub.execute_input":"2023-03-13T06:42:24.378194Z","iopub.status.idle":"2023-03-13T06:42:24.442905Z","shell.execute_reply.started":"2023-03-13T06:42:24.378150Z","shell.execute_reply":"2023-03-13T06:42:24.441566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Util Functions","metadata":{}},{"cell_type":"code","source":"def openAudioFile(path, sample_rate=44100, as_mono=True, mean_substract=False):\n    \n    # Open file with librosa\n    sig, rate = librosa.load(path, sr=sample_rate, mono=as_mono)\n\n    # Noise reduction?\n    if mean_substract:\n        sig -= sig.mean()\n\n    return sig, rate","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:24.444310Z","iopub.execute_input":"2023-03-13T06:42:24.444682Z","iopub.status.idle":"2023-03-13T06:42:33.829173Z","shell.execute_reply.started":"2023-03-13T06:42:24.444646Z","shell.execute_reply":"2023-03-13T06:42:33.827723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def splitSignal(sig, rate, seconds, overlap, minlen):\n\n    # Split signal with overlap\n    sig_splits = []\n    for i in range(0, len(sig), int((seconds - overlap) * rate)):\n        split = sig[i:i + int(seconds * rate)]\n\n        # End of signal?\n        if len(split) < int(minlen * rate):\n            break\n        \n        # Signal chunk too short?\n        if len(split) < int(rate * seconds):\n            split = np.hstack((split, np.zeros((int(rate * seconds) - len(split),))))\n        \n        sig_splits.append(split)\n\n    return sig_splits","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def melspec(sig, rate, shape=(128, 256), fmin=500, fmax=15000, normalize=True, preemphasis=0.95):\n\n    # shape = (height, width) in pixels\n\n    # Mel-Spec parameters\n    SAMPLE_RATE = rate\n    N_FFT = shape[0] * 8 # = window length\n    N_MELS = shape[0]\n    HOP_LEN = len(sig) // (shape[1] - 1)\n    print('hop len = ' + str(HOP_LEN))    \n    FMAX = fmax\n    FMIN = fmin\n\n    # Preemphasis as in python_speech_features by James Lyons\n    if preemphasis:\n        sig = np.append(sig[0], sig[1:] - preemphasis * sig[:-1])\n\n    \n    # Librosa mel-spectrum\n    melspec = librosa.feature.melspectrogram(y=sig, sr=SAMPLE_RATE, hop_length=HOP_LEN, n_fft=N_FFT, n_mels=N_MELS, fmax=FMAX, fmin=FMIN, power=1.0)\n    \n    # Convert power spec to dB scale (compute dB relative to peak power)\n    melspec = librosa.amplitude_to_db(melspec, ref=np.max, top_db=80)\n\n    # Flip spectrum vertically (only for better visialization, low freq. at bottom)\n    melspec = melspec[::-1, ...]\n    #print(melspec.shape)\n\n    # Trim to desired shape if too large\n    melspec = melspec[:shape[0], :shape[1]]\n\n    # Normalize values between 0 and 1\n    if normalize:\n        melspec -= melspec.min()\n        if not melspec.max() == 0:\n            melspec /= melspec.max()\n        else:\n            melspec = np.clip(melspec, 0, 1)\n\n    return melspec.astype('float32')\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## To be continued...","metadata":{}},{"cell_type":"code","source":"# Find the name of the class with the top score when mean-aggregated across frames.\ndef class_names_from_csv(class_map_csv_text):\n    \"\"\"Returns list of class names corresponding to score vector.\"\"\"\n    with open(labels_path) as csv_file:\n        csv_reader = csv.reader(csv_file, delimiter=',')\n        class_names = [mid for mid, desc in csv_reader]\n        return class_names[1:]\n\n## note that the bird classifier classifies a much larger set of birds than the\n## competition, so we need to load the model's set of class names or else our \n## indices will be off.\nclasses = class_names_from_csv(labels_path)","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:33.830544Z","iopub.execute_input":"2023-03-13T06:42:33.830895Z","iopub.status.idle":"2023-03-13T06:42:33.850389Z","shell.execute_reply.started":"2023-03-13T06:42:33.830862Z","shell.execute_reply":"2023-03-13T06:42:33.848988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_metadata = pd.read_csv(\"/kaggle/input/birdclef-2023/train_metadata.csv\")\ntrain_metadata.head()\ncompetition_classes = sorted(train_metadata.primary_label.unique())\n\nforced_defaults = 0\ncompetition_class_map = []\nfor c in competition_classes:\n    try:\n        i = classes.index(c)\n        competition_class_map.append(i)\n    except:\n        competition_class_map.append(0)\n        forced_defaults += 1\n        \n## this is the count of classes not supported by our pretrained model\n## you could choose to simply not predict these, set a default as above,\n## or create your own model using the pretrained model as a base.\nforced_defaults","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:33.852309Z","iopub.execute_input":"2023-03-13T06:42:33.852941Z","iopub.status.idle":"2023-03-13T06:42:34.032193Z","shell.execute_reply.started":"2023-03-13T06:42:33.852898Z","shell.execute_reply":"2023-03-13T06:42:34.030476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 4: Preprocess the data\n\nThe following functions are one way to load the audio provided and break it up into the five-second samples with a sample rate of 32,000 required by the competition.","metadata":{}},{"cell_type":"code","source":"def frame_audio(\n      audio_array: np.ndarray,\n      window_size_s: float = 5.0,\n      hop_size_s: float = 5.0,\n      sample_rate = 32000,\n      ) -> np.ndarray:\n    \n    \"\"\"Helper function for framing audio for inference.\"\"\"\n    \"\"\" using tf.signal \"\"\"\n    if window_size_s is None or window_size_s < 0:\n        return audio_array[np.newaxis, :]\n    frame_length = int(window_size_s * sample_rate)\n    hop_length = int(hop_size_s * sample_rate)\n    framed_audio = tf.signal.frame(audio_array, frame_length, hop_length, pad_end=True)\n    return framed_audio\n\ndef ensure_sample_rate(waveform, original_sample_rate,\n                       desired_sample_rate=32000):\n    \"\"\"Resample waveform if required.\"\"\"\n    if original_sample_rate != desired_sample_rate:\n        waveform = tfio.audio.resample(waveform, original_sample_rate, desired_sample_rate)\n    return desired_sample_rate, waveform","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:34.033688Z","iopub.execute_input":"2023-03-13T06:42:34.034073Z","iopub.status.idle":"2023-03-13T06:42:34.043504Z","shell.execute_reply.started":"2023-03-13T06:42:34.034025Z","shell.execute_reply":"2023-03-13T06:42:34.041712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Below we load one training sample - use the Audio function to listen to the samples inside the notebook!","metadata":{}},{"cell_type":"code","source":"audio, sample_rate = librosa.load(\"/kaggle/input/birdclef-2023/train_audio/afghor1/XC156639.ogg\")\nsample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\nAudio(wav_data, rate=sample_rate)","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:34.046182Z","iopub.execute_input":"2023-03-13T06:42:34.046751Z","iopub.status.idle":"2023-03-13T06:42:34.825177Z","shell.execute_reply.started":"2023-03-13T06:42:34.046696Z","shell.execute_reply":"2023-03-13T06:42:34.823973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 5: Make predictions\n\nEach test sample is cut into 5-second chunks. We use the pretrained model to return probabilities for all 10k birds included in the model, then pull out the classes used in this competition to create a final submission row. Note that we are NOT doing anything special to handle the 3 missing classes; those will need fine-tuning / transfer learning, which will be handled in a separate notebook.","metadata":{}},{"cell_type":"code","source":"fixed_tm = frame_audio(wav_data)\nlogits, embeddings = model.infer_tf(fixed_tm[:1])\nprobabilities = tf.nn.softmax(logits)\nargmax = np.argmax(probabilities)\nprint(f\"The audio is from the class {classes[argmax]} (element:{argmax} in the label.csv file), with probability of {probabilities[0][argmax]}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:34.826310Z","iopub.execute_input":"2023-03-13T06:42:34.826675Z","iopub.status.idle":"2023-03-13T06:42:45.753358Z","shell.execute_reply.started":"2023-03-13T06:42:34.826637Z","shell.execute_reply":"2023-03-13T06:42:45.751850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_for_sample(filename, sample_submission, frame_limit_secs=None):\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    \n    audio, sample_rate = librosa.load(filename)\n    sample_rate, wav_data = ensure_sample_rate(audio, sample_rate)\n    \n    fixed_tm = frame_audio(wav_data)\n    \n    frame = 5\n    all_logits, all_embeddings = model.infer_tf(fixed_tm[:1])\n    for window in fixed_tm[1:]:\n        if frame_limit_secs and frame > frame_limit_secs:\n            continue\n        \n        logits, embeddings = model.infer_tf(window[np.newaxis, :])\n        all_logits = np.concatenate([all_logits, logits], axis=0)\n        frame += 5\n    \n    frame = 5\n    all_probabilities = []\n    for frame_logits in all_logits:\n        probabilities = tf.nn.softmax(frame_logits).numpy()\n        \n        ## set the appropriate row in the sample submission\n        sample_submission.loc[sample_submission.row_id == file_id + \"_\" + str(frame), competition_classes] = probabilities[competition_class_map]\n        frame += 5","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:45.757017Z","iopub.execute_input":"2023-03-13T06:42:45.757697Z","iopub.status.idle":"2023-03-13T06:42:45.766446Z","shell.execute_reply.started":"2023-03-13T06:42:45.757655Z","shell.execute_reply":"2023-03-13T06:42:45.765396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Step 6: Generate a submission\n\nNow we process all of the test samples as discussed above, creating output rows, and saving them in the provided `sample_submission.csv`. Finally, we save these rows to our final output file: `submission.csv`. This is the file that gets submitted and scored when you submit the notebook.","metadata":{}},{"cell_type":"code","source":"test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\ntest_samples","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:45.767911Z","iopub.execute_input":"2023-03-13T06:42:45.768560Z","iopub.status.idle":"2023-03-13T06:42:45.791771Z","shell.execute_reply.started":"2023-03-13T06:42:45.768520Z","shell.execute_reply":"2023-03-13T06:42:45.790327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub = pd.read_csv(\"/kaggle/input/birdclef-2023/sample_submission.csv\")\nsample_sub[competition_classes] = sample_sub[competition_classes].astype(np.float32)\nsample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:45.793743Z","iopub.execute_input":"2023-03-13T06:42:45.794550Z","iopub.status.idle":"2023-03-13T06:42:45.912226Z","shell.execute_reply.started":"2023-03-13T06:42:45.794497Z","shell.execute_reply":"2023-03-13T06:42:45.910903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"frame_limit_secs = 15 if sample_sub.shape[0] == 3 else None\nfor sample_filename in test_samples:\n    predict_for_sample(sample_filename, sample_sub, frame_limit_secs=15)","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:45.913976Z","iopub.execute_input":"2023-03-13T06:42:45.915151Z","iopub.status.idle":"2023-03-13T06:42:57.725046Z","shell.execute_reply.started":"2023-03-13T06:42:45.915097Z","shell.execute_reply":"2023-03-13T06:42:57.723891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:57.726796Z","iopub.execute_input":"2023-03-13T06:42:57.727200Z","iopub.status.idle":"2023-03-13T06:42:57.757460Z","shell.execute_reply.started":"2023-03-13T06:42:57.727158Z","shell.execute_reply":"2023-03-13T06:42:57.756356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-13T06:42:57.759378Z","iopub.execute_input":"2023-03-13T06:42:57.760429Z","iopub.status.idle":"2023-03-13T06:42:57.782627Z","shell.execute_reply.started":"2023-03-13T06:42:57.760377Z","shell.execute_reply":"2023-03-13T06:42:57.780945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}