{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Import Libraries ","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_io as tfio\n\nimport math, random\nimport os\nimport csv\nimport cv2\nimport glob\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom pathlib import Path\n\nimport librosa\nimport torch\nimport torchaudio\nfrom torchvision.models import resnet34\nfrom torchaudio import transforms\nfrom IPython.display import Audio\n\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-08T00:16:24.046513Z","iopub.execute_input":"2023-05-08T00:16:24.047530Z","iopub.status.idle":"2023-05-08T00:16:38.301510Z","shell.execute_reply.started":"2023-05-08T00:16:24.047434Z","shell.execute_reply":"2023-05-08T00:16:38.299765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Configuration","metadata":{}},{"cell_type":"code","source":"class Config:\n    device = \"tpu\"\n    \n    base_path = \"/kaggle/input/birdclef-2023\"\n    train_metadata = \"/train_metadata.csv\"\n    train_audio = \"/train_audio\"\n    test_audio = \"/test_soundscapes\"\n    taxonomies = \"/eBird_Taxonomy_v2021.csv\"\n    sample_submission = \"/sample_submission.csv\"\n    \n    audio_length = 5 # seconds\n    sample_rate = 32000\n    image_size = (512, 512)\n    \n    # Mel Spectrogram\n    mels = 30\n    fmin = 0\n    fmax = 8000\n    n_fft= 700\n    hop_length = 548\n    \n    # Masking\n    max_mask_pct = 0.1\n    n_freq_masks = 1\n    n_time_masks = 1\n    \n    batch_size = 16\n    \n    epochs = 4","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:38.305242Z","iopub.execute_input":"2023-05-08T00:16:38.306933Z","iopub.status.idle":"2023-05-08T00:16:38.316331Z","shell.execute_reply.started":"2023-05-08T00:16:38.306886Z","shell.execute_reply":"2023-05-08T00:16:38.314798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# resolver = tf.distribute.cluster_resolver.TPUClusterResolver(tpu='')\n# tf.config.experimental_connect_to_cluster(resolver)\n# # This is the TPU initialization code that has to be at the beginning.\n# tf.tpu.experimental.initialize_tpu_system(resolver)\n# print(\"All devices: \", tf.config.list_logical_devices('TPU'))\n\n# strategy = tf.distribute.TPUStrategy(resolver)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:38.317908Z","iopub.execute_input":"2023-05-08T00:16:38.318923Z","iopub.status.idle":"2023-05-08T00:16:38.332730Z","shell.execute_reply.started":"2023-05-08T00:16:38.318869Z","shell.execute_reply":"2023-05-08T00:16:38.331392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import the Metadata Dataframe","metadata":{}},{"cell_type":"code","source":"# Define the metadata path\nmetadata_path = Path(Config.base_path + Config.train_metadata)\n\ndf = pd.read_csv(metadata_path)\ndf[\"filename\"] = df[\"filename\"].apply(lambda x: Config.base_path + Config.train_audio + \"/\" + x)\n# df['filename'] = df['filename'].str.replace('.ogg', '.wav')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:38.336416Z","iopub.execute_input":"2023-05-08T00:16:38.336965Z","iopub.status.idle":"2023-05-08T00:16:38.529201Z","shell.execute_reply.started":"2023-05-08T00:16:38.336904Z","shell.execute_reply":"2023-05-08T00:16:38.527961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.utils import resample\ntarget_size = math.ceil(df[\"primary_label\"].value_counts().mean())\n\ndf_balanced = pd.DataFrame()\n\n\nfor class_label in df[\"primary_label\"].unique():\n    resampled_df = resample(df[df[\"primary_label\"] == class_label], n_samples=target_size)\n    df_balanced = pd.concat([\n        df_balanced, \n        resampled_df\n    ], ignore_index=True)\n    \n    \ndf = df_balanced","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:38.530372Z","iopub.execute_input":"2023-05-08T00:16:38.530701Z","iopub.status.idle":"2023-05-08T00:16:39.647148Z","shell.execute_reply.started":"2023-05-08T00:16:38.530670Z","shell.execute_reply":"2023-05-08T00:16:39.646066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OneHotEncoder\n\nenc = OneHotEncoder(sparse=False)\n\n# cats = df[\"primary_label\"]\nencoded_classes = enc.fit_transform(df[['primary_label']].values.reshape(-1, 1))\nclass_names = enc.get_feature_names_out(['class'])\nclass_df = pd.DataFrame(encoded_classes, columns=class_names)\nclass_df[class_names] = class_df[class_names].astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:39.648598Z","iopub.execute_input":"2023-05-08T00:16:39.648934Z","iopub.status.idle":"2023-05-08T00:16:40.985906Z","shell.execute_reply.started":"2023-05-08T00:16:39.648901Z","shell.execute_reply":"2023-05-08T00:16:40.984588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.concat([\n    df,\n    class_df\n], axis=1)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:40.988343Z","iopub.execute_input":"2023-05-08T00:16:40.988857Z","iopub.status.idle":"2023-05-08T00:16:41.049108Z","shell.execute_reply.started":"2023-05-08T00:16:40.988800Z","shell.execute_reply":"2023-05-08T00:16:41.047786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"figure = plt.figure(figsize=[15, 15])\nplt.bar(df[\"primary_label\"].unique(), df[\"primary_label\"].value_counts())\nplt.title(\"Bird Class Distribution\")","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:41.050658Z","iopub.execute_input":"2023-05-08T00:16:41.051029Z","iopub.status.idle":"2023-05-08T00:16:45.257447Z","shell.execute_reply.started":"2023-05-08T00:16:41.050986Z","shell.execute_reply":"2023-05-08T00:16:45.256188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels_path = hub.resolve('https://kaggle.com/models/google/bird-vocalization-classifier/frameworks/tensorFlow2/variations/bird-vocalization-classifier/versions/1') + \"/assets/label.csv\"","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:45.260781Z","iopub.execute_input":"2023-05-08T00:16:45.261557Z","iopub.status.idle":"2023-05-08T00:16:45.270090Z","shell.execute_reply.started":"2023-05-08T00:16:45.261505Z","shell.execute_reply":"2023-05-08T00:16:45.268894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Credit @PHIL CULLITON https://www.kaggle.com/code/philculliton/inferring-birds-with-kaggle-models\n\n# Find the name of the class with the top score when mean-aggregated across frames.\ndef class_names_from_csv(class_map_csv_text):\n    \"\"\"Returns list of class names corresponding to score vector.\"\"\"\n    with open(labels_path) as csv_file:\n        csv_reader = csv.reader(csv_file, delimiter=',')\n        class_names = [mid for mid, desc in csv_reader]\n        return class_names[1:]\n\n## note that the bird classifier classifies a much larger set of birds than the\n## competition, so we need to load the model's set of class names or else our \n## indices will be off.\nclasses = class_names_from_csv(labels_path)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:45.275310Z","iopub.execute_input":"2023-05-08T00:16:45.275688Z","iopub.status.idle":"2023-05-08T00:16:45.305159Z","shell.execute_reply.started":"2023-05-08T00:16:45.275653Z","shell.execute_reply":"2023-05-08T00:16:45.303897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_classes = sorted(df.primary_label.unique())\nnew_class = len(classes)\ncompetition_class_map = {}\nunseen_classes = []\nfor c in competition_classes:\n    try:\n        i = classes.index(c)\n        competition_class_map[c] = i\n    except:\n        competition_class_map[c] = new_class\n        unseen_classes.append(c)\n        new_class += 1        \n\n        \n## this is the count of classes not supported by our pretrained model\n## you could choose to simply not predict these, set a default as above,\n## or create your own model using the pretrained model as a base.\nunseen_classes","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:45.306551Z","iopub.execute_input":"2023-05-08T00:16:45.306921Z","iopub.status.idle":"2023-05-08T00:16:45.351963Z","shell.execute_reply.started":"2023-05-08T00:16:45.306885Z","shell.execute_reply":"2023-05-08T00:16:45.350622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define the data preprocessing functions\nAudio must be:\n- Loaded as a tensor\n- Rechanneled if needed\n- Resampled\n- Set to proper length (Padding/truncation)\n- Time Shifted\n- Turned into a Mel Spectrogram\n- Spectro Time Masking\n- Spectro Frequency Masking","metadata":{}},{"cell_type":"code","source":"class AudioUtils():\n    \"\"\"\n    Loading and preprocessing\n    \"\"\"\n    @staticmethod\n    def openFile(filepath, label):\n        \"\"\"\n        Reads a filepath and returns a lazy-loaded IOtensor.\n        \"\"\"\n        audio_tensor = tf.io.read_file(filepath)\n        audio_tensor = tfio.audio.decode_vorbis(audio_tensor)\n        audio_tensor = tf.cast(audio_tensor, dtype=tf.float32) / 32768.0 \n        audio_tensor = tf.squeeze(audio_tensor, axis=[-1])\n        return audio_tensor, label\n    \n    @staticmethod\n    def playAudio(t, label):\n        \"\"\"\n        Use Ipython to preview the audio sample\n        \"\"\"\n        length = int(tf.shape(t)[0].numpy())\n        \n#         audio_tensor = tf.squeeze(t, axis=[-1])\n        audio_tensor = t\n        return Audio(audio_tensor.numpy(), rate=32000)\n    \n    @staticmethod\n    def visualizeAudio(t, label):\n        \"\"\"\n        Visualize the audio in waveform\n        \"\"\"\n        # Convert it to a float32 tensor\n        visual_tensor = tf.squeeze(t, axis=[-1])\n        visual_tensor = tf.cast(t, tf.float32) / 32768.0\n        \n        # Plot\n        plt.figure()\n        plt.plot(visual_tensor.numpy())\n    \n    @staticmethod\n    def resizeTensor(t, label, sr=32000, desired_length=Config.audio_length, random_start=False):\n        \"\"\"\n        the length of the audio clip to the desired length\n        \"\"\"\n        tensor_size = int(tf.shape(t)[0].numpy())\n        if tensor_size > (desired_length * sr):\n            resized_audio_tensor = tf.image.random_crop(t, size=[sr * Config.audio_length, ]) # Random crop to 5 seconds\n            \n        else:\n            # Find the missing length to fill the tensor\n            rem = (desired_length * sr) - tensor_size\n            # Create a start padding and end padding to add to the tensor\n            front_pad, back_pad = tf.zeros(math.floor(rem/2)), tf.zeros(math.ceil(rem/2))\n            # Resize \n            resized_audio_tensor = tf.concat([front_pad, t, back_pad], 0)\n        \n        return resized_audio_tensor, label\n    \n    @staticmethod\n    def resample(t, label, input_rate=32000, desired_rate=Config.sample_rate):\n        if (input_rate == desired_rate):\n            return t, label\n        resampled_tensor = tfio.audio.resample(t, rate_in=input_rate, rate_out=desired_rate)\n        resampled_tensor = tf.squeeze(resampled_tensor, axis=[-1])\n        return resampled_tensor, label\n    \n    \"\"\"\n    Data Augmentations\n    \"\"\"\n    @staticmethod\n    def shiftTime(t, label, shift_limit=0.4):\n        \"\"\"\n        Roll the time randomly\n        \"\"\"\n        sig_len = int(tf.shape(t)[0].numpy())\n        shift_amt = int(random.random() * shift_limit * sig_len)\n        t = tf.roll(t, shift_amt, 0)\n        return t, label\n    \n    @staticmethod\n    def createMelSpectrogram(t, label, sample_rate=Config.sample_rate, nfft=Config.n_fft, mels=Config.mels, fmin=Config.fmin, fmax=Config.fmax, top_db=80):\n        \"\"\"\n        Create a mel spectrogram by scaling the frequency and the amplitude \n        \"\"\"\n        spectrogram = tfio.audio.spectrogram(\n            t, nfft=nfft, window=nfft, stride=300)\n        # Melscale the frequency\n        mel_spectrogram = tfio.audio.melscale(\n            spectrogram, sample_rate, 40, fmin, fmax\n        )\n        # Mel scale the amplitude\n        mel_spectrogram = tfio.audio.dbscale(\n            mel_spectrogram, top_db=top_db\n        )\n#         mel_spectrogram = tf.reshape(mel_spectrogram, [534, 40, 1])\n#         resized_mel_spectrogram = tf.image.resize(mel_spectrogram, [*Config.image_size])\n#         resized_mel_spectrogram = tf.repeat(resized_mel_spectrogram, 3, axis=-1)\n        return mel_spectrogram, label\n        \n        \n    @staticmethod\n    def spectroMasking(spectro, label, max_mask_pct=Config.max_mask_pct, n_freq_masks=Config.n_freq_masks, n_time_masks=Config.n_time_masks):\n        \"\"\"\n        Apply masking methods to the time and frequency axes of the mel spectrogram\n        \"\"\"\n        n_mels, n_steps = 40, 300\n        aug_spec = spectro\n        \n        freq_mask_param = max_mask_pct * n_mels\n        for _ in range(n_freq_masks):\n            aug_spec = tfio.audio.freq_mask(aug_spec, param=2)\n            \n        time_mask_param = int(max_mask_pct * n_steps)\n        for _ in range(n_time_masks):\n            aug_spec = tfio.audio.time_mask(aug_spec, param=2)\n        \n        return aug_spec, label\n    \n    @staticmethod\n    def resizeFinalTensor(spectro, label):\n        resized_mel_spectrogram = tf.reshape(spectro, [534, 40, 1])\n        resized_mel_spectrogram = tf.image.resize(resized_mel_spectrogram, [*Config.image_size])\n        resized_mel_spectrogram = tf.repeat(resized_mel_spectrogram, 3, axis=-1)\n        return resized_mel_spectrogram, label\n    \n    @staticmethod\n    def normalize(spectro, label):\n        \"\"\"\n        Normalize the values to between 0 and 1\n        \"\"\"\n        max_value = tf.reduce_max(spectro)\n        normalized_tensor = spectro / max_value\n        return normalized_tensor, label\n    \n    \n    @staticmethod\n    def visualizeSpectro(spectro, label):\n        img = spectro.numpy()\n        img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        \n        plt.figure()\n#         plt.imshow(img.transpose([1, 0]), cmap='gray')\n        plt.imshow(img)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:19:59.393425Z","iopub.execute_input":"2023-05-08T00:19:59.394051Z","iopub.status.idle":"2023-05-08T00:19:59.419472Z","shell.execute_reply.started":"2023-05-08T00:19:59.394006Z","shell.execute_reply":"2023-05-08T00:19:59.418201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# rand = random.randrange(0, len(df) - 1)\n# example = df.iloc[rand, -265:]\n# class_name = df.loc[rand, \"primary_label\"]\n# filepath = example[\"filename\"]\n# example_label = list(example.iloc[-264:].values)\n\n# audio, label = AudioUtils.openFile(filepath, example_label)\n# audio, label = AudioUtils.resizeTensor(audio, label)\n\n# display(class_name)\n# AudioUtils.playAudio(audio, label)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:45.381715Z","iopub.execute_input":"2023-05-08T00:16:45.382106Z","iopub.status.idle":"2023-05-08T00:16:45.394546Z","shell.execute_reply.started":"2023-05-08T00:16:45.382061Z","shell.execute_reply":"2023-05-08T00:16:45.393419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\nfilenames, labels = df.loc[:, \"filename\"].values, df.loc[:, class_names].values\nfile_dataset = tf.data.Dataset.from_tensor_slices(filenames.astype(bytes))\nlabels_dataset = tf.data.Dataset.from_tensor_slices(labels)\nlabels_dataset = labels_dataset.map(lambda x: tf.cast(x, dtype=tf.int32), num_parallel_calls=AUTO)\ndataset = tf.data.Dataset.zip((file_dataset, labels_dataset))\ndataset = dataset.shuffle(2048)\ndataset = dataset.map(lambda x, y: (AudioUtils.openFile(x, y)), num_parallel_calls=AUTO)\ndataset = dataset.map(lambda x, y: (tf.py_function(func=AudioUtils.resizeTensor, \n                                                   inp=[x, y], \n                                                   Tout=[tf.float32,tf.int32])))\ndataset = dataset.map(lambda x, y: (AudioUtils.resample(x, y)), num_parallel_calls=AUTO)\ndataset = dataset.map(lambda x, y: (tf.py_function(func=AudioUtils.shiftTime, \n                                                   inp=[x, y], \n                                                   Tout=[tf.float32,tf.int32])))\n\ndataset = dataset.map(lambda x, y: (tf.py_function(func=AudioUtils.createMelSpectrogram, \n                                                   inp=[x, y], \n                                                   Tout=[tf.float32,tf.int32])))\ndataset = dataset.map(lambda x, y: (tf.py_function(func=AudioUtils.spectroMasking, \n                                                   inp=[x, y], \n                                                   Tout=[tf.float32,tf.int32])))\ndataset = dataset.map(lambda x, y: (AudioUtils.resizeFinalTensor(x, y)))\ndataset = dataset.map(lambda x, y: (AudioUtils.normalize(x, y)))\ndataset = dataset.batch(Config.batch_size)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:45.396526Z","iopub.execute_input":"2023-05-08T00:16:45.396883Z","iopub.status.idle":"2023-05-08T00:16:46.598221Z","shell.execute_reply.started":"2023-05-08T00:16:45.396852Z","shell.execute_reply":"2023-05-08T00:16:46.596821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def _fixup_shape(images, labels):\n    images.set_shape([None, None, None, 3])\n    labels.set_shape([None, len(class_names)])\n    return images, labels\n\ndataset = dataset.map(_fixup_shape)\ndataset = dataset.prefetch(buffer_size=AUTO)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:46.599892Z","iopub.execute_input":"2023-05-08T00:16:46.600458Z","iopub.status.idle":"2023-05-08T00:16:46.663972Z","shell.execute_reply.started":"2023-05-08T00:16:46.600407Z","shell.execute_reply":"2023-05-08T00:16:46.662749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for x, y in dataset.take(1):\n    example, ex_label = x[0], y[0]\n    AudioUtils.visualizeSpectro(example, ex_label)\n    break\n    \n# AudioUtils.visualizeSpectro(example, ex_label)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:16:46.665509Z","iopub.execute_input":"2023-05-08T00:16:46.665959Z","iopub.status.idle":"2023-05-08T00:16:49.760977Z","shell.execute_reply.started":"2023-05-08T00:16:46.665922Z","shell.execute_reply":"2023-05-08T00:16:49.759748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_file, sample_label = filenames[200], labels[200]\nutil = AudioUtils()\n\naudio, label = util.openFile(sample_file, sample_label)\nresized_audio, label = util.resizeTensor(audio, label)\nresampled_audio, label = util.resample(resized_audio, label)\nmel_spectrogram, label = util.createMelSpectrogram(resampled_audio, label)\nresized_mel, label = util.resizeFinalTensor(mel_spectrogram, label)\nnormalized_mel, label = util.normalize(resized_mel, label)\nutil.visualizeSpectro(normalized_mel, label)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:20:29.685604Z","iopub.execute_input":"2023-05-08T00:20:29.686002Z","iopub.status.idle":"2023-05-08T00:20:30.120451Z","shell.execute_reply.started":"2023-05-08T00:20:29.685968Z","shell.execute_reply":"2023-05-08T00:20:30.119245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"masked_mel_spectrogram, label = util.spectroMasking(mel_spectrogram, label)\nresized_mel, label = util.resizeFinalTensor(masked_mel_spectrogram, label)\nnormalized_mel, label = util.normalize(resized_mel, label)\nutil.visualizeSpectro(normalized_mel, label)","metadata":{"execution":{"iopub.status.busy":"2023-05-08T00:20:30.122683Z","iopub.execute_input":"2023-05-08T00:20:30.123442Z","iopub.status.idle":"2023-05-08T00:20:30.395460Z","shell.execute_reply.started":"2023-05-08T00:20:30.123390Z","shell.execute_reply":"2023-05-08T00:20:30.394254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train, Test, Val Split","metadata":{}},{"cell_type":"code","source":"# Credit @Angel Igareta https://towardsdatascience.com/how-to-split-a-tensorflow-dataset-into-train-validation-and-test-sets-526c8dd29438\ndef get_dataset_partitions_tf(ds, ds_size, train_split=0.8, val_split=0.1, test_split=0.1, shuffle=False, shuffle_size=10000):\n    assert (train_split + test_split + val_split) == 1\n    \n    if shuffle:\n        # Specify seed to always have the same split distribution between runs\n        ds = ds.shuffle(shuffle_size, seed=12)\n    \n    train_size = int(train_split * ds_size)\n    val_size = int(val_split * ds_size)\n    \n    train_ds = ds.take(train_size)    \n    val_ds = ds.skip(train_size).take(val_size)\n    test_ds = ds.skip(train_size).skip(val_size)\n    \n    return train_ds, val_ds, test_ds","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:57:56.258813Z","iopub.execute_input":"2023-05-06T19:57:56.259208Z","iopub.status.idle":"2023-05-06T19:57:56.269826Z","shell.execute_reply.started":"2023-05-06T19:57:56.259164Z","shell.execute_reply":"2023-05-06T19:57:56.268744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds, val_ds, test_ds = get_dataset_partitions_tf(dataset, len(dataset))","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:57:56.271650Z","iopub.execute_input":"2023-05-06T19:57:56.272479Z","iopub.status.idle":"2023-05-06T19:57:56.299077Z","shell.execute_reply.started":"2023-05-06T19:57:56.272438Z","shell.execute_reply":"2023-05-06T19:57:56.297700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for x,y in train_ds:\n    print(tf.shape(x), tf.shape(y))\n    break","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:57:56.301075Z","iopub.execute_input":"2023-05-06T19:57:56.301482Z","iopub.status.idle":"2023-05-06T19:57:57.940827Z","shell.execute_reply.started":"2023-05-06T19:57:56.301446Z","shell.execute_reply":"2023-05-06T19:57:57.939732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load and Train the EfficientNetB0 Model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import layers\nfrom tensorflow.keras.models import Sequential\n\ngpus = tf.config.list_logical_devices('GPU')\nstrategy = tf.distribute.MirroredStrategy(gpus)\nNUM_CLASSES = len(class_names)\nwith strategy.scope():\n    inputs = layers.Input(shape=(*Config.image_size, 3))\n    outputs = tf.keras.applications.efficientnet.EfficientNetB0(include_top=True, \n                                                                weights=None, \n                                                                classes=NUM_CLASSES, \n                                                                input_shape=(*Config.image_size, 3))(inputs)\n    model = tf.keras.Model(inputs, outputs)\n    model.compile(\n        optimizer=\"adam\", loss=\"categorical_crossentropy\", metrics=['accuracy']\n    )\n    \n# Define the ModelCheckpoint callback with a filepath that includes the epoch number\ncheckpoints = tf.keras.callbacks.ModelCheckpoint(filepath='/kaggle/working/256-px_wmasking_run_1_weights.h5', \n                                                  monitor='val_loss', \n                                                  save_weights_only=True,\n                                                  mode='auto')\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:57:57.942487Z","iopub.execute_input":"2023-05-06T19:57:57.942962Z","iopub.status.idle":"2023-05-06T19:58:03.173517Z","shell.execute_reply.started":"2023-05-06T19:57:57.942920Z","shell.execute_reply":"2023-05-06T19:58:03.172450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.load_weights(\"/kaggle/input/256-px-run-2-weights/256-px_run_2_weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:58:03.174999Z","iopub.execute_input":"2023-05-06T19:58:03.177968Z","iopub.status.idle":"2023-05-06T19:58:04.492709Z","shell.execute_reply.started":"2023-05-06T19:58:03.177924Z","shell.execute_reply":"2023-05-06T19:58:04.491495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\nepochs = 14\nstart = time.time()\nhist = model.fit(train_ds, epochs=epochs, validation_data=val_ds, verbose=1, callbacks=[checkpoints])\nend  = time.time()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T19:58:04.494175Z","iopub.execute_input":"2023-05-06T19:58:04.495537Z","iopub.status.idle":"2023-05-07T04:15:02.896172Z","shell.execute_reply.started":"2023-05-06T19:58:04.495496Z","shell.execute_reply":"2023-05-07T04:15:02.895141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save_weights(\"/kaggle/working/256-px_run_3_weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2023-05-07T04:15:02.899065Z","iopub.execute_input":"2023-05-07T04:15:02.899417Z","iopub.status.idle":"2023-05-07T04:15:03.327272Z","shell.execute_reply.started":"2023-05-07T04:15:02.899387Z","shell.execute_reply":"2023-05-07T04:15:03.325803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"duration = (end-start) / 60\nduration","metadata":{"execution":{"iopub.status.busy":"2023-05-07T04:15:03.329493Z","iopub.execute_input":"2023-05-07T04:15:03.330318Z","iopub.status.idle":"2023-05-07T04:15:03.346579Z","shell.execute_reply.started":"2023-05-07T04:15:03.330263Z","shell.execute_reply":"2023-05-07T04:15:03.344251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define a callback to update the checkpoint filepath with the epoch number\nclass UpdateCheckpointPathCallback(tf.keras.callbacks.Callback):\n    def on_epoch_end(self, epoch, logs=None):\n        # Update the checkpoint filepath with the epoch number\n        new_filepath = f'/kaggle/working/epoch_{epoch+1:02d}_saved_weights.h5'\n        self.model.checkpoints.filepath = new_filepath\n\n# Instantiate the callback\nupdate_checkpoint_path_callback = UpdateCheckpointPathCallback()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T04:15:03.349652Z","iopub.execute_input":"2023-05-07T04:15:03.350412Z","iopub.status.idle":"2023-05-07T04:15:03.363311Z","shell.execute_reply.started":"2023-05-07T04:15:03.350371Z","shell.execute_reply":"2023-05-07T04:15:03.362091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n\ndef plot_hist(hist):\n    plt.plot(hist.history[\"accuracy\"])\n    plt.plot(hist.history[\"val_accuracy\"])\n    plt.title(\"model accuracy\")\n    plt.ylabel(\"accuracy\")\n    plt.xlabel(\"epoch\")\n    plt.legend([\"train\", \"validation\"], loc=\"upper left\")\n    plt.show()\n\n\nplot_hist(hist)","metadata":{"execution":{"iopub.status.busy":"2023-05-07T04:15:03.365347Z","iopub.execute_input":"2023-05-07T04:15:03.366120Z","iopub.status.idle":"2023-05-07T04:15:03.592228Z","shell.execute_reply.started":"2023-05-07T04:15:03.366081Z","shell.execute_reply":"2023-05-07T04:15:03.591221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Custom Train Loader\n_______________________________________________________________","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers.experimental import preprocessing\n\nmodel = tf.keras.applications.efficientnet.EfficientNetB0(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(224, 224, 3),\n    drop_connect_rate=0.4,\n)\n\ninput_shape = (224, 224, 3)\n# Define the pre-processing layers for the input spectrogram\npreprocessing_layers = [\n    preprocessing.Normalization(),\n]\n\n# Define the spectrogram classification model\nmodel = tf.keras.Sequential([\n    tf.keras.Input(shape=input_shape),\n    *preprocessing_layers,\n    model,\n    tf.keras.layers.GlobalAveragePooling2D(),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(len(class_names), activation='softmax')\n])\nmodel.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2023-03-22T22:37:01.495739Z","iopub.execute_input":"2023-03-22T22:37:01.496134Z","iopub.status.idle":"2023-03-22T22:37:01.507667Z","shell.execute_reply.started":"2023-03-22T22:37:01.496102Z","shell.execute_reply":"2023-03-22T22:37:01.506445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss_object = tf.keras.losses.categorical_crossentropy# Define loss and grad\ndef loss_fn(model, x, y, training):\n    # training=training is needed only if there are layers with different\n    # behavior during training versus inference (e.g. Dropout).\n    y_ = model(x, training=training)\n    return loss_object(y_true=y, y_pred=y_)\n\ndef grad(model, inputs, targets):\n    with tf.GradientTape() as tape:\n        loss_value = loss_fn(model, inputs, targets, training=True)\n    return loss_value, tape.gradient(loss_value, model.trainable_variables)","metadata":{"execution":{"iopub.status.busy":"2023-03-22T22:27:30.046246Z","iopub.execute_input":"2023-03-22T22:27:30.047230Z","iopub.status.idle":"2023-03-22T22:27:30.054254Z","shell.execute_reply.started":"2023-03-22T22:27:30.047181Z","shell.execute_reply":"2023-03-22T22:27:30.053039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for audio, label in train_ds:\n    print(audio.shape)\n    predictions = model(audio, training=True)","metadata":{"execution":{"iopub.status.busy":"2023-03-22T22:27:30.493956Z","iopub.execute_input":"2023-03-22T22:27:30.494628Z","iopub.status.idle":"2023-03-22T22:28:12.024605Z","shell.execute_reply.started":"2023-03-22T22:27:30.494592Z","shell.execute_reply":"2023-03-22T22:28:12.022362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def trainModel(train_ds, val_ds, test_ds=None, epochs=Config.epochs):\n#     device = tf.device('/GPU:0')\n    train_loss_results = []\n    train_accuracy_results = []\n    val_loss_results = []\n    val_accuracy_results = []\n    class_predictions = [] \n    predictions = []\n\n    optimizer = tf.keras.optimizers.SGD(learning_rate=0.01)\n    for epoch in range(epochs):\n        print(f'Training on epoch #{epoch+1}')\n        epoch_loss_avg = tf.keras.metrics.Mean()\n        epoch_accuracy = tf.keras.metrics.CategoricalAccuracy()\n\n        for audio, label in train_ds:\n            predictions = model(audio, training=True)\n            loss, grads = grad(model, audio, label)\n            optimizer.apply_gradients(zip(grads, model.trainable_variables))\n\n            epoch_loss_avg.update_state(loss)\n            epoch_accuracy.update_state(label, predictions)\n\n        train_loss_results.append(epoch_loss_avg)\n        train_accuracy_results.append(epoch_accuracy)\n\n        print(f'\\n Loss: {train_loss_results[-1].result().numpy()}, Accuracy: {train_accuracy_results[-1].result().numpy()}')\n\n        val_loss_avg = tf.keras.metrics.Mean()\n        val_accuracy = tf.keras.metrics.CategoricalAccuracy()\n\n        for audio, label in val_ds:\n            predictions = model(audio, training=False)\n            loss = loss_fn(model, audio, label, training=False)\n\n            val_loss_avg.update_state(loss)\n            val_accuracy.update_state(label, predictions)\n\n        val_loss_results.append(val_loss_avg)\n        val_accuracy_results.append(val_accuracy)\n\n        print(f'\\n Val Loss: {val_loss_results[-1].result().numpy()}, Val Accuracy: {val_accuracy_results[-1].result().numpy()}')\n\n        if test_ds is not None:\n        # Use the model to predict on the testing data\n            for audio, label in test_ds:\n                predictions.append([model.predict(audio), label])\n    return class_predictions","metadata":{"execution":{"iopub.status.busy":"2023-03-22T18:56:50.017732Z","iopub.execute_input":"2023-03-22T18:56:50.018666Z","iopub.status.idle":"2023-03-22T18:56:50.030660Z","shell.execute_reply.started":"2023-03-22T18:56:50.018615Z","shell.execute_reply":"2023-03-22T18:56:50.029659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainModel(train_ds, val_ds)","metadata":{"execution":{"iopub.status.busy":"2023-03-22T19:11:46.140305Z","iopub.execute_input":"2023-03-22T19:11:46.141249Z","iopub.status.idle":"2023-03-22T19:11:59.113983Z","shell.execute_reply.started":"2023-03-22T19:11:46.141195Z","shell.execute_reply":"2023-03-22T19:11:59.112634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare the data for predictions from the pretrained model","metadata":{}},{"cell_type":"code","source":"def frameAudio(audio, window=float(5.0), hop_length=float(5.0), sample_rate=Config.sample_rate):\n    if window is None or window < 0:\n        return audio[np.newaxis, :]\n    \n    frame_length = int(window * sample_rate)\n    hop_length = int(hop_length * sample_rate)\n    framed_audio = tf.signal.frame(audio, frame_length, hop_length, pad_end=True)\n    return framed_audio\n\ndef ensureSampleRate(audio, sample_rate, desired_sample_rate=32000):\n    if sample_rate != desired_sample_rate:\n        audio = tfio.audio.resample(audio, sample_rate, desired_sample_rate)\n    return audio, desired_sample_rate","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:58:01.136245Z","iopub.execute_input":"2023-03-21T03:58:01.137669Z","iopub.status.idle":"2023-03-21T03:58:01.146368Z","shell.execute_reply.started":"2023-03-21T03:58:01.137607Z","shell.execute_reply":"2023-03-21T03:58:01.144743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_for_sample(filename, submission_df):\n    file_id = filename.split(\".ogg\")[0].split(\"/\")[-1]\n    \n    audio, sample_rate = librosa.load(filename)\n    wav_data, sample_rate = ensureSampleRate(audio, sample_rate)\n    \n    fixed_tm = frameAudio(wav_data)\n    \n    frame = 5\n    all_logits, all_embeddings = model.infer_tf(fixed_tm[:1])\n    for window in fixed_tm[1:]:\n        \n        logits, embeddings = model.infer_tf(window[np.newaxis, :])\n        all_logits = np.concatenate([all_logits, logits], axis=0)\n        frame += 5\n        if frame > 10:\n            break\n    \n    frame = 5\n    all_probabilities = []\n    for frame_logits in all_logits:\n        probabilities = tf.nn.softmax(frame_logits).numpy()\n        \n        ## set the appropriate row in the sample submission\n        submission_df.loc[submission_df.row_id == file_id + \"_\" + str(frame), competition_classes] = probabilities[list(competition_class_map.values())]\n        frame += 5","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:58:01.276974Z","iopub.execute_input":"2023-03-21T03:58:01.278032Z","iopub.status.idle":"2023-03-21T03:58:01.288210Z","shell.execute_reply.started":"2023-03-21T03:58:01.277984Z","shell.execute_reply":"2023-03-21T03:58:01.286643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Build submission Dataframe and Predict the test data","metadata":{}},{"cell_type":"code","source":"submission_df = pd.read_csv(Config.base_path + Config.sample_submission)\nsubmission_df[competition_classes] = submission_df[competition_classes].astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:58:02.440568Z","iopub.execute_input":"2023-03-21T03:58:02.440998Z","iopub.status.idle":"2023-03-21T03:58:02.520917Z","shell.execute_reply.started":"2023-03-21T03:58:02.440964Z","shell.execute_reply":"2023-03-21T03:58:02.519840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_samples = list(glob.glob(\"/kaggle/input/birdclef-2023/test_soundscapes/*.ogg\"))\ntest_samples","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:58:05.343134Z","iopub.execute_input":"2023-03-21T03:58:05.345024Z","iopub.status.idle":"2023-03-21T03:58:05.359329Z","shell.execute_reply.started":"2023-03-21T03:58:05.344958Z","shell.execute_reply":"2023-03-21T03:58:05.357567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for sample in test_samples:\n    predict_for_sample(sample, submission_df)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:53:39.244227Z","iopub.execute_input":"2023-03-21T03:53:39.244656Z","iopub.status.idle":"2023-03-21T03:53:49.593231Z","shell.execute_reply.started":"2023-03-21T03:53:39.244619Z","shell.execute_reply":"2023-03-21T03:53:49.591995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-03-21T03:49:55.252450Z","iopub.status.idle":"2023-03-21T03:49:55.252907Z","shell.execute_reply.started":"2023-03-21T03:49:55.252683Z","shell.execute_reply":"2023-03-21T03:49:55.252705Z"},"trusted":true},"execution_count":null,"outputs":[]}]}