{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# BirdCLEF 2023 🐦\n> Identify bird calls in soundscapes\n\n<img src=\"https://storage.googleapis.com/kaggle-competitions/kaggle/44224/logos/header.png?t=2023-03-06-18-30-53\">","metadata":{}},{"cell_type":"markdown","source":"# Methodology  🎯\n* This notebook will demonstrate **Bird Call Identification** with `TensorFlow`. \n* This notebook will also show how to infer using `TensorFlow`. For training check below mentioned notebook.\n* This notebook will use `5sec` audio recording as per requirements. But training is done on much more larger size recording. Dynamic shape is utilize to infer on a different resolution.\n* This notebook will consider one recording as one batch which will speed up the processing.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import sys, os\nsys.path.append('/kaggle/input/efficientnet-keras-dataset/efficientnet_kaggle')","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:33.511974Z","iopub.execute_input":"2023-03-15T12:28:33.512385Z","iopub.status.idle":"2023-03-15T12:28:33.519346Z","shell.execute_reply.started":"2023-03-15T12:28:33.512347Z","shell.execute_reply":"2023-03-15T12:28:33.517557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Libraries 📚","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\ntf.get_logger().setLevel('ERROR')\ntf.autograph.set_verbosity(0)\nimport os\nimport pandas as pd\nimport numpy as np\nimport random\nfrom glob import glob\nfrom tqdm import tqdm\ntqdm.pandas()\nimport gc\nimport librosa\nimport sklearn\nimport time\n\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\nimport librosa.display as lid\nimport IPython.display as ipd\n\nimport tensorflow as tf\ntf.config.optimizer.set_jit(True) # enable xla for speed up\nimport tensorflow_io as tfio\nimport tensorflow.keras.backend as K\n\nimport efficientnet.tfkeras as efn","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:33.522281Z","iopub.execute_input":"2023-03-15T12:28:33.522943Z","iopub.status.idle":"2023-03-15T12:28:33.539449Z","shell.execute_reply.started":"2023-03-15T12:28:33.522883Z","shell.execute_reply":"2023-03-15T12:28:33.538270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Library version","metadata":{}},{"cell_type":"code","source":"print('np:', np.__version__)\nprint('pd:', pd.__version__)\nprint('sklearn:', sklearn.__version__)\nprint('librosa:', librosa.__version__)\nprint('tf:', tf.__version__)\nprint('tfio:', tfio.__version__)","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:33.541809Z","iopub.execute_input":"2023-03-15T12:28:33.542723Z","iopub.status.idle":"2023-03-15T12:28:33.553707Z","shell.execute_reply.started":"2023-03-15T12:28:33.542681Z","shell.execute_reply":"2023-03-15T12:28:33.551861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Configuration ⚙️","metadata":{}},{"cell_type":"code","source":"class CFG:\n    debug = False\n    verbose = 0\n    \n    device = 'CPU'\n    seed = 42\n    \n    # Input image size and batch size\n    img_size = [128, 384]\n    batch_size = 16\n    infer_bs = 2\n    tta = 1\n    drop_remainder = True\n    \n    # STFT parameters\n    duration = 5 # duration for test\n    train_duration = 10\n    sample_rate = 32000\n    downsample = 1\n    trim = True\n    audio_len = duration*sample_rate\n    nfft = 2028\n    window = 2048\n    hop_length = train_duration*32000 // (img_size[1] - 1)\n    fmin = 20\n    fmax = 16000\n    normalize = True\n\n    # Data Preprocessing Settings\n    class_names = sorted(os.listdir('/kaggle/input/birdclef-2023/train_audio/'))\n    num_classes = len(class_names)\n    class_labels = list(range(num_classes))\n    label2name = dict(zip(class_labels, class_names))\n    name2label = {v:k for k,v in label2name.items()}\n    \n    target_col = ['target']\n    tab_cols = ['filename','common_name','rate']","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:33.555004Z","iopub.execute_input":"2023-03-15T12:28:33.555513Z","iopub.status.idle":"2023-03-15T12:28:33.569423Z","shell.execute_reply.started":"2023-03-15T12:28:33.555438Z","shell.execute_reply":"2023-03-15T12:28:33.567692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def seeding(SEED):\n    np.random.seed(SEED)\n    random.seed(SEED)\n    os.environ['PYTHONHASHSEED'] = str(SEED)\n#     os.environ['TF_CUDNN_DETERMINISTIC'] = str(SEED)\n    tf.random.set_seed(SEED)\n    print('seeding done!!!')\nseeding(CFG.seed)","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:33.572997Z","iopub.execute_input":"2023-03-15T12:28:33.573594Z","iopub.status.idle":"2023-03-15T12:28:33.584126Z","shell.execute_reply.started":"2023-03-15T12:28:33.573520Z","shell.execute_reply":"2023-03-15T12:28:33.582301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Setting up device 📱","metadata":{}},{"cell_type":"code","source":"def get_device():\n    \"Detect and intializes GPU/TPU automatically\"\n    # Check TPU category\n    tpu = 'local' if CFG.device=='TPU-VM' else None\n    try:\n        # Connect to TPU\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect(tpu=tpu) \n        # Set TPU strategy\n        strategy = tf.distribute.TPUStrategy(tpu)\n        print(f'> Running on {CFG.device} ', tpu.master(), end=' | ')\n        print('Num of TPUs: ', strategy.num_replicas_in_sync)\n        device=CFG.device\n    except:\n        # If TPU is not available, detect GPUs\n        gpus = tf.config.list_logical_devices('GPU')\n        ngpu = len(gpus)\n         # Check number of GPUs\n        if ngpu:\n            # Set GPU strategy\n            strategy = tf.distribute.MirroredStrategy(gpus) # single-GPU or multi-GPU\n            # Print GPU details\n            print(\"> Running on GPU\", end=' | ')\n            print(\"Num of GPUs: \", ngpu)\n            device='GPU'\n        else:\n            # If no GPUs are available, use CPU\n            print(\"> Running on CPU\")\n            strategy = tf.distribute.get_strategy()\n            device='CPU'\n    return strategy, device, tpu","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:33.586132Z","iopub.execute_input":"2023-03-15T12:28:33.586723Z","iopub.status.idle":"2023-03-15T12:28:33.611722Z","shell.execute_reply.started":"2023-03-15T12:28:33.586678Z","shell.execute_reply":"2023-03-15T12:28:33.609931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize GPU/TPU/TPU-VM\nstrategy, CFG.device, tpu = get_device()\nCFG.replicas = strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:33.613778Z","iopub.execute_input":"2023-03-15T12:28:33.614424Z","iopub.status.idle":"2023-03-15T12:28:33.628245Z","shell.execute_reply.started":"2023-03-15T12:28:33.614379Z","shell.execute_reply":"2023-03-15T12:28:33.626691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset Path 📖","metadata":{}},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/birdclef-2023'\nGCS_PATH = BASE_PATH","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:33.630974Z","iopub.execute_input":"2023-03-15T12:28:33.631529Z","iopub.status.idle":"2023-03-15T12:28:33.640336Z","shell.execute_reply.started":"2023-03-15T12:28:33.631474Z","shell.execute_reply":"2023-03-15T12:28:33.639041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Meta Data 📖","metadata":{}},{"cell_type":"code","source":"test_paths = glob('/kaggle/input/birdclef-2023/test_soundscapes/*ogg')\ntest_df = pd.DataFrame(test_paths, columns=['filepath'])\ntest_df['filename'] = test_df.filepath.map(lambda x: x.split('/')[-1].replace('.ogg',''))\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:33.730702Z","iopub.execute_input":"2023-03-15T12:28:33.732084Z","iopub.status.idle":"2023-03-15T12:28:33.752006Z","shell.execute_reply.started":"2023-03-15T12:28:33.732030Z","shell.execute_reply":"2023-03-15T12:28:33.750035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.io.gfile.exists(test_df.filepath.iloc[0])","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:33.754844Z","iopub.execute_input":"2023-03-15T12:28:33.755312Z","iopub.status.idle":"2023-03-15T12:28:33.765676Z","shell.execute_reply.started":"2023-03-15T12:28:33.755272Z","shell.execute_reply":"2023-03-15T12:28:33.763884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Loader ","metadata":{}},{"cell_type":"code","source":"def load_audio(filepath, sr=32000, normalize=True):\n    audio, orig_sr = librosa.load(filepath, sr=None)\n    if sr!=orig_sr:\n        audio = librosa.resample(y, orig_sr, sr)\n    audio = audio.astype('float32').ravel()\n    audio = tf.convert_to_tensor(audio)\n    if normalize:\n        audio = Normalize(audio)\n    return audio\n\n# @tf.function\n# def load_audio_tf(filepath, normalize=True):\n#     file_bytes = tf.io.read_file(filepath)\n#     audio = tfio.audio.decode_vorbis(file_bytes)\n# #     audio = tf.cast(audio_file.to_tensor, tf.float32)\n#     audio = tf.squeeze(audio, axis=-1)\n#     if normalize:\n#         audio = Normalize(audio)\n#     return audio\n\n# Standardize the audio\n@tf.function(jit_compile=True)\ndef Normalize(data, min_max=True):\n    # Compute the mean and standard deviation of the data\n    MEAN = tf.math.reduce_mean(data)\n    STD = tf.math.reduce_std(data)\n    # Standardize the data\n    data = tf.math.divide_no_nan(data - MEAN, STD)\n    # Normalize to [0, 1]\n    if min_max:\n        MIN = tf.math.reduce_min(data)\n        MAX = tf.math.reduce_max(data)\n        data = tf.math.divide_no_nan(data - MIN, MAX - MIN)\n    return data\n\n@tf.function(jit_compile=True)\ndef Spec2Img(spec):\n    spec = tf.tile(spec[..., tf.newaxis], [1, 1, 1, 3])\n    return spec\n\n@tf.function(jit_compile=True)\ndef Img2Spec(img):\n    return img[..., 0]\n\n@tf.function(jit_compile=True)\ndef MakeFrame(audio, duration=5, sr=32000):\n    frame_length = int(duration * sr)\n    frame_step = int(duration * sr)\n    chunks = tf.signal.frame(audio, frame_length, frame_step, pad_end=True)\n    return chunks\n\n@tf.function(jit_compile=True)\ndef Audio2Spec(audio, spec_shape = CFG.img_size, sr=CFG.sample_rate, \n                    nfft=CFG.nfft, window=CFG.window, fmin=CFG.fmin, fmax=CFG.fmax, return_img=True):\n    spec_height = spec_shape[0]\n    spec_width = spec_shape[1]\n    hop_length = tf.cast(CFG.hop_length, tf.int32) # sample rate * duration / spec width - 1 == 627\n    spec = tfio.audio.spectrogram(audio, nfft=nfft, window=window, stride=hop_length)\n    mel_spec = tfio.audio.melscale(spec, rate=sr, mels=spec_height, fmin=fmin, fmax=fmax)\n    db_mel_spec = tfio.audio.dbscale(mel_spec, top_db=80)\n    db_mel_spec = tf.linalg.matrix_transpose(db_mel_spec) # to keep it (batch, mel, time)\n    return db_mel_spec","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:33.767863Z","iopub.execute_input":"2023-03-15T12:28:33.768510Z","iopub.status.idle":"2023-03-15T12:28:33.806892Z","shell.execute_reply.started":"2023-03-15T12:28:33.768447Z","shell.execute_reply":"2023-03-15T12:28:33.805490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA🎨","metadata":{}},{"cell_type":"code","source":"def display_audio(row):\n    # Caption for viz\n    caption = f'Id: {row.filename}'\n    # Read audio file\n    audio = load_audio(row.filepath)\n    # Keep fixed length audio\n    audio = audio[:CFG.audio_len]\n    # Spectrogram from audio\n    spec = Audio2Spec(audio, return_img=False)\n    # Display audio\n    print(\"# Audio:\")\n    display(ipd.Audio(audio.numpy(), rate=CFG.sample_rate))\n    print('# Visualization:')\n    fig, ax = plt.subplots(2, 1, figsize=(12, 2*3), sharex=True, tight_layout=True)\n    fig.suptitle(caption)\n    # Waveplot\n    lid.waveshow(audio.numpy(),\n                 sr=CFG.sample_rate,\n                 ax=ax[0])\n    # Specplot\n    lid.specshow(spec.numpy(), \n                 sr = CFG.sample_rate, \n                 hop_length = CFG.hop_length,\n                 n_fft=CFG.nfft,\n                 fmin=CFG.fmin,\n                 fmax=CFG.fmax,\n                 x_axis = 'time', \n                 y_axis = 'mel',\n                 cmap = 'coolwarm',\n                 ax=ax[1])\n    ax[0].set_xlabel('');\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:33.812251Z","iopub.execute_input":"2023-03-15T12:28:33.812673Z","iopub.status.idle":"2023-03-15T12:28:33.824205Z","shell.execute_reply.started":"2023-03-15T12:28:33.812637Z","shell.execute_reply":"2023-03-15T12:28:33.822548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display_audio(test_df.iloc[0])","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:33.826351Z","iopub.execute_input":"2023-03-15T12:28:33.826838Z","iopub.status.idle":"2023-03-15T12:28:37.617357Z","shell.execute_reply.started":"2023-03-15T12:28:33.826785Z","shell.execute_reply":"2023-03-15T12:28:37.615603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference Configs 🔧","metadata":{}},{"cell_type":"code","source":"# Directory of checkpoint\nCKPT_DIR = '/kaggle/input/birdclef23-effnet-fsr-cutmixup-train-ds'\n# Get file paths of all trained models in the directory\nCKPT_PATHS = sorted(glob(f'{CKPT_DIR}/*h5'))\n# Load all the models in memory to speed up\nCKPTS = [tf.keras.models.load_model(x, compile=False) for x in tqdm(CKPT_PATHS, desc=\"Loading ckpts \")]\n# Num of ckpt to use\nNUM_CKPTS = 1\n\n# Submit or Interactive mode\nSUBMIT = pd.read_csv('/kaggle/input/birdclef-2023/sample_submission.csv').shape[0] != 3","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:37.619046Z","iopub.execute_input":"2023-03-15T12:28:37.619439Z","iopub.status.idle":"2023-03-15T12:28:37.647921Z","shell.execute_reply.started":"2023-03-15T12:28:37.619401Z","shell.execute_reply":"2023-03-15T12:28:37.646477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"# Start stopwatch\ntick = time.time()\n\n# Initialize empty list to store ids\nids = []\n# Initialize empty array to store predictions\npreds = np.empty(shape=(0, 264), dtype='float32')\n\n# Iterate over each audio file in the test dataset\nfor filepath in tqdm(test_df.filepath.tolist(), 'test '):\n    # Extract the filename without the extension\n    filename = filepath.split('/')[-1].replace('.ogg','')\n    \n    # Load audio from file and create audio frames, each recording will be a batch input\n    audio = load_audio(filepath)\n    chunks = MakeFrame(audio)\n    \n#     # If not submitting, only use the first three frames for speed\n#     if not SUBMIT:\n#         chunks = chunks[:3]\n    \n    # Convert audio frames to spectrograms + rgb image using a vectorized function \n    specs = Audio2Spec(chunks)\n    specs = Spec2Img(specs)\n    \n    # Predict bird species for all frames in a recording using all trained models\n    chunk_preds = np.zeros(shape=(len(specs), 264), dtype=np.float32)\n    for model in CKPTS[:NUM_CKPTS]:\n        # Get the model's predictions for the current audio frames\n        rec_preds = model(specs, training=False).numpy()\n        # Ensemble all prediction with average\n        chunk_preds += rec_preds/len(CKPTS)\n    \n    # Create a ID for each frame in a recording using the filename and frame number\n    rec_ids = [f'{filename}_{(frame_id+1)*5}' for frame_id in range(len(chunks))]\n    \n    # Concatenate the ids\n    ids += rec_ids\n    # Concatenate the predictions\n    preds = np.concatenate([preds, chunk_preds], axis=0)\n    \n# Stop stopwatch\ntock = time.time()","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:37.652151Z","iopub.execute_input":"2023-03-15T12:28:37.652621Z","iopub.status.idle":"2023-03-15T12:28:48.770563Z","shell.execute_reply.started":"2023-03-15T12:28:37.652581Z","shell.execute_reply":"2023-03-15T12:28:48.768235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submissions 📮","metadata":{}},{"cell_type":"code","source":"# Submit prediction\npred_df = pd.DataFrame(ids, columns=['row_id'])\npred_df.loc[:, CFG.class_names] = preds\npred_df.to_csv('submission.csv',index=False)\npred_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:48.773816Z","iopub.execute_input":"2023-03-15T12:28:48.774222Z","iopub.status.idle":"2023-03-15T12:28:48.872624Z","shell.execute_reply.started":"2023-03-15T12:28:48.774187Z","shell.execute_reply":"2023-03-15T12:28:48.871195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Check Submission ","metadata":{}},{"cell_type":"code","source":"if not SUBMIT:\n    pred_labels = pred_df[pred_df.columns[1:]].values.argmax(axis=1)\n    pred_classes = list(map(lambda x: CFG.label2name[x], pred_labels))\n    print(pred_classes)","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:48.874443Z","iopub.execute_input":"2023-03-15T12:28:48.875286Z","iopub.status.idle":"2023-03-15T12:28:48.884959Z","shell.execute_reply.started":"2023-03-15T12:28:48.875233Z","shell.execute_reply":"2023-03-15T12:28:48.883315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission Time ⏰\nEstimated time to complete the submission.\n> **Note**: There are nearly ~$200$ recordings on the test data.","metadata":{}},{"cell_type":"code","source":"sub_time = (tock-tick)*200 # ~200 recording on the test data\nsub_time = time.gmtime(sub_time)\nsub_time = time.strftime(\"%H hr: %M min : %S sec\", sub_time)\nprint(f\">> Time for submission: ~ {sub_time}\")","metadata":{"execution":{"iopub.status.busy":"2023-03-15T12:28:48.886708Z","iopub.execute_input":"2023-03-15T12:28:48.888435Z","iopub.status.idle":"2023-03-15T12:28:48.899818Z","shell.execute_reply.started":"2023-03-15T12:28:48.888387Z","shell.execute_reply":"2023-03-15T12:28:48.897979Z"},"trusted":true},"execution_count":null,"outputs":[]}]}