{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":182219678,"sourceType":"kernelVersion"}],"dockerImageVersionId":30732,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nos.environ[\"KERAS_BACKEND\"] = \"tensorflow\"  # \"jax\" or \"tensorflow\" or \"torch\" \n\nimport keras_cv\nimport keras\nimport keras.backend as K\nimport tensorflow as tf\nimport tensorflow_io as tfio\n\nimport numpy as np \nimport pandas as pd\n\nfrom glob import glob\nfrom tqdm import tqdm\n\nimport librosa\nimport IPython.display as ipd\nimport librosa.display as lid\n\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\n\ncmap = mpl.cm.get_cmap('coolwarm')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-09T02:14:39.548000Z","iopub.execute_input":"2024-06-09T02:14:39.548536Z","iopub.status.idle":"2024-06-09T02:15:06.436853Z","shell.execute_reply.started":"2024-06-09T02:14:39.548489Z","shell.execute_reply":"2024-06-09T02:15:06.435455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class CFG:\n    seed = 42\n    \n    # Input image size and batch size\n    img_size = [128, 384]\n    \n    # Audio duration, sample rate, and length\n    duration = 15 # second\n    sample_rate = 32000\n    audio_len = duration*sample_rate\n    \n    # STFT parameters\n    nfft = 2028\n    window = 2048\n    hop_length = audio_len // (img_size[1] - 1)\n    fmin = 20\n    fmax = 16000\n    \n    # Number of epochs, model name\n    preset = 'efficientnetv2_b2_imagenet'\n\n    # Class Labels for BirdCLEF 24\n    class_names = sorted(os.listdir('/kaggle/input/birdclef-2024/train_audio/'))\n    num_classes = len(class_names)\n    class_labels = list(range(num_classes))\n    label2name = dict(zip(class_labels, class_names))\n    name2label = {v:k for k,v in label2name.items()}","metadata":{"execution":{"iopub.status.busy":"2024-06-09T02:15:06.439147Z","iopub.execute_input":"2024-06-09T02:15:06.440434Z","iopub.status.idle":"2024-06-09T02:15:06.462017Z","shell.execute_reply.started":"2024-06-09T02:15:06.440390Z","shell.execute_reply":"2024-06-09T02:15:06.460349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.set_random_seed(CFG.seed)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T02:15:06.464053Z","iopub.execute_input":"2024-06-09T02:15:06.464446Z","iopub.status.idle":"2024-06-09T02:15:06.487230Z","shell.execute_reply.started":"2024-06-09T02:15:06.464413Z","shell.execute_reply":"2024-06-09T02:15:06.485477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_PATH = '/kaggle/input/birdclef-2024'","metadata":{"execution":{"iopub.status.busy":"2024-06-09T02:15:06.490974Z","iopub.execute_input":"2024-06-09T02:15:06.491501Z","iopub.status.idle":"2024-06-09T02:15:06.501537Z","shell.execute_reply.started":"2024-06-09T02:15:06.491454Z","shell.execute_reply":"2024-06-09T02:15:06.500086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_paths = glob(f'{BASE_PATH}/test_soundscapes/*ogg')\n# During commit use `unlabeled` data as there is no `test` data.\n# During submission `test` data will automatically be populated.\nif len(test_paths)==0:\n    test_paths = glob(f'{BASE_PATH}/unlabeled_soundscapes/*ogg')[:10]\ntest_df = pd.DataFrame(test_paths, columns=['filepath'])\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T02:15:06.503204Z","iopub.execute_input":"2024-06-09T02:15:06.503707Z","iopub.status.idle":"2024-06-09T02:15:06.755747Z","shell.execute_reply.started":"2024-06-09T02:15:06.503662Z","shell.execute_reply":"2024-06-09T02:15:06.754440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras import Model\nfrom keras import layers\n\ndef naive_inception_module(layer_in, f1, f2, f3):\n    # 5x9 conv\n    conv1_1 = layers.Conv2D(f1, (1,1), padding='same', activation=layers.LeakyReLU(negative_slope=0.25),strides=4,kernel_initializer='random_normal',bias_initializer='zeros')(layer_in)\n    # 7x7 conv\n    conv1_2 = layers.Conv2D(f2, (3,3), padding='same', activation=layers.LeakyReLU(negative_slope=0.25),strides=4,kernel_initializer='random_normal',bias_initializer='zeros')(layer_in)\n    # 9x5 conv\n    conv1_3 = layers.Conv2D(f3, (5,5), padding='same', activation=layers.LeakyReLU(negative_slope=0.25),strides=4,kernel_initializer='random_normal',bias_initializer='zeros')(layer_in)\n    \n    # concatenate filters, assumes filters/channels last\n    layer_out = layers.concatenate([conv1_1, conv1_2, conv1_3], axis=-1)\n    return layer_out\n \n# define model2 input\nvisible = layers.Input(shape=(128, 384, 3))\n# add inception module\nlayer = naive_inception_module(visible, 4, 4, 4)\n\nconv2=layers.Conv2D(filters=16, kernel_size=(1,1),padding='valid', activation=layers.LeakyReLU(negative_slope=0.25),strides=1,kernel_initializer='random_normal',bias_initializer='zeros')(layers.BatchNormalization()(layer))\nmaxpool2=layers.MaxPooling2D(pool_size=(2, 2),strides=1)(conv2)\nnorm2=(layers.BatchNormalization()(maxpool2))\n\nconv3=layers.Conv2D(filters=32, kernel_size=(3,3),padding='valid', \n                    activation=layers.LeakyReLU(negative_slope=0.25),strides=2,kernel_initializer='random_normal',bias_initializer='ones')(norm2)\nmaxpool3=layers.MaxPooling2D(pool_size=(2, 2),strides=1)(conv3)\nnorm3=(layers.BatchNormalization()(maxpool3))\n\nconv4=layers.Conv2D(filters=64, kernel_size=(3,3),padding='valid', \n                    activation=layers.LeakyReLU(negative_slope=0.25),strides=2,kernel_initializer='random_normal',bias_initializer='ones')(norm3)\nmaxpool4=layers.MaxPooling2D(pool_size=(2, 2),strides=1)(conv4)\nnorm4=(layers.BatchNormalization()(maxpool4))\n\nconv5=layers.Conv2D(filters=64, kernel_size=(5,5),padding='valid', \n                    activation=layers.LeakyReLU(negative_slope=0.25),strides=2,kernel_initializer='random_normal',bias_initializer='ones')(norm4)\nmaxpool5=layers.MaxPooling2D(pool_size=(2, 2),strides=1)(conv5)\nnorm5=(layers.BatchNormalization()(conv5))\n\nflat=layers.Flatten()(norm5)\n\nFC1=layers.Dense(64, activation=layers.LeakyReLU(negative_slope=0.25),kernel_initializer='random_normal',bias_initializer='ones')(flat)\ndrop_FC1=layers.Dropout((0.5))(layers.BatchNormalization()(FC1))\nFC2=layers.Dense(32, activation=layers.LeakyReLU(negative_slope=0.25),kernel_initializer='random_normal',bias_initializer='ones')(drop_FC1)\ndrop_FC2=drop3=layers.Dropout((0.5))(layers.BatchNormalization()(FC2))\n\nFC3=layers.Dense(16, activation=layers.LeakyReLU(negative_slope=0.25),kernel_initializer='random_normal',bias_initializer='ones')(drop_FC2)\ndrop_FC3=drop3=layers.Dropout((0.5))(layers.BatchNormalization()(FC3))\n\n\nlayer_out=layers.Dense(CFG.num_classes, activation='softmax',kernel_initializer='random_normal',bias_initializer='zeros')(drop_FC3)\n\n# create model2\nmodel2 = Model(inputs=visible, outputs=layer_out)\n\nmodel2.load_weights(\"/kaggle/input/birdclefakcnn/AKCNN_model1.weights.h5\")\n","metadata":{"execution":{"iopub.status.busy":"2024-06-09T02:15:06.757260Z","iopub.execute_input":"2024-06-09T02:15:06.757654Z","iopub.status.idle":"2024-06-09T02:15:07.243111Z","shell.execute_reply.started":"2024-06-09T02:15:06.757623Z","shell.execute_reply":"2024-06-09T02:15:07.241776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Decodes Audio\ndef build_decoder(with_labels=True, dim=1024):\n    def get_audio(filepath):\n        file_bytes = tf.io.read_file(filepath)\n        audio = tfio.audio.decode_vorbis(file_bytes) # decode .ogg file\n        audio = tf.cast(audio, tf.float32)\n        if tf.shape(audio)[1]>1: # stereo -> mono\n            audio = audio[...,0:1]\n        audio = tf.squeeze(audio, axis=-1)\n        return audio\n    \n    def create_frames(audio, duration=5, sr=32000):\n        frame_size = int(duration * sr)\n        audio = tf.pad(audio[..., None], [[0, tf.shape(audio)[0] % frame_size], [0, 0]]) # pad the end\n        audio = tf.squeeze(audio) # remove extra dimension added for padding\n        frames = tf.reshape(audio, [-1, frame_size]) # shape: [num_frames, frame_size]\n        return frames\n    \n    def apply_preproc(spec):\n        # Standardize\n        mean = tf.math.reduce_mean(spec)\n        std = tf.math.reduce_std(spec)\n        spec = tf.where(tf.math.equal(std, 0), spec - mean, (spec - mean) / std)\n\n        # Normalize using Min-Max\n        min_val = tf.math.reduce_min(spec)\n        max_val = tf.math.reduce_max(spec)\n        spec = tf.where(tf.math.equal(max_val - min_val, 0), spec - min_val,\n                              (spec - min_val) / (max_val - min_val))\n        return spec\n\n    def decode(path):\n        # Load audio file\n        audio = get_audio(path)\n        # Split audio file into frames with each having 5 seecond duration\n        audio = create_frames(audio)\n        # Convert audio to spectrogram\n        spec = keras.layers.MelSpectrogram(num_mel_bins=CFG.img_size[0],\n                                             fft_length=CFG.nfft, \n                                              sequence_stride=CFG.hop_length, \n                                              sampling_rate=CFG.sample_rate)(audio)\n        # Apply normalization and standardization\n        spec = apply_preproc(spec)\n        # Covnert spectrogram to 3 channel image (for imagenet)\n        spec = tf.tile(spec[..., None], [1, 1, 1, 3])\n        return spec\n    \n    return decode","metadata":{"execution":{"iopub.status.busy":"2024-06-09T02:15:07.245040Z","iopub.execute_input":"2024-06-09T02:15:07.245559Z","iopub.status.idle":"2024-06-09T02:15:07.263766Z","shell.execute_reply.started":"2024-06-09T02:15:07.245495Z","shell.execute_reply":"2024-06-09T02:15:07.262383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build data loader\ndef build_dataset(paths, batch_size=1, decode_fn=None, cache=False):\n    if decode_fn is None:\n        decode_fn = build_decoder(dim=CFG.audio_len) # decoder\n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = (paths,)\n    ds = tf.data.Dataset.from_tensor_slices(slices)\n    ds = ds.map(decode_fn, num_parallel_calls=AUTO) # decode audio to spectrograms then create frames\n    ds = ds.cache() if cache else ds # cache files\n    ds = ds.batch(batch_size, drop_remainder=False) # create batches\n    ds = ds.prefetch(AUTO)\n    return ds","metadata":{"execution":{"iopub.status.busy":"2024-06-09T02:15:07.265123Z","iopub.execute_input":"2024-06-09T02:15:07.265594Z","iopub.status.idle":"2024-06-09T02:15:07.285851Z","shell.execute_reply.started":"2024-06-09T02:15:07.265545Z","shell.execute_reply":"2024-06-09T02:15:07.284555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize empty list to store ids\nids = []\n\n# Initialize empty array to store predictions\npreds = np.empty(shape=(0, CFG.num_classes), dtype='float32')\n\n# Build test dataset\ntest_paths = test_df.filepath.tolist()\ntest_ds = build_dataset(paths=test_paths, batch_size=1)\n\n# Iterate over each audio file in the test dataset\nfor idx, specs in enumerate(tqdm(iter(test_ds), desc='test ', total=len(test_df))):\n    # Extract the filename without the extension\n    filename = test_paths[idx].split('/')[-1].replace('.ogg','')\n    \n    specs = keras.ops.convert_to_tensor(specs[0])\n\n    # Resize the input tensor to the expected shape\n    resized_specs = tf.image.resize(specs, [128, 384])\n\n    # Predict bird species for all frames in a recording using all trained models\n    frame_preds = model2.predict(resized_specs, verbose=0)\n\n    # Create an ID for each frame in a recording using the filename and frame number\n    frame_ids = [f'{filename}_{(frame_id+1)*5}' for frame_id in range(len(frame_preds))]\n    \n    # Concatenate the ids\n    ids += frame_ids\n    # Concatenate the predictions\n    preds = np.concatenate([preds, frame_preds], axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-06-09T02:19:39.941373Z","iopub.execute_input":"2024-06-09T02:19:39.941831Z","iopub.status.idle":"2024-06-09T02:19:50.723042Z","shell.execute_reply.started":"2024-06-09T02:19:39.941795Z","shell.execute_reply":"2024-06-09T02:19:50.721768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Submit prediction\npred_df = pd.DataFrame(ids, columns=['row_id'])\npred_df.loc[:, CFG.class_names] = preds\npred_df.to_csv('submission.csv',index=False)\npred_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-09T02:20:10.382149Z","iopub.execute_input":"2024-06-09T02:20:10.382637Z","iopub.status.idle":"2024-06-09T02:20:10.698084Z","shell.execute_reply.started":"2024-06-09T02:20:10.382601Z","shell.execute_reply":"2024-06-09T02:20:10.696909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}