{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\n\nimport warnings\nwarnings.filterwarnings(action='ignore')\n\nimport pandas as pd\nimport librosa\nimport numpy as np\n\nfrom sklearn.utils import shuffle\nfrom PIL import Image\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\n\n# Global vars\nRANDOM_SEED = 1337\nSAMPLE_RATE = 32000\nSIGNAL_LENGTH = 5 # seconds\nSPEC_SHAPE = (48, 128) # height x width\nFMIN = 500\nFMAX = 12500\nMAX_AUDIO_FILES = 1500","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:37:36.641665Z","iopub.execute_input":"2021-06-08T09:37:36.64209Z","iopub.status.idle":"2021-06-08T09:37:44.724765Z","shell.execute_reply.started":"2021-06-08T09:37:36.642008Z","shell.execute_reply":"2021-06-08T09:37:44.723456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Code adapted from: \n# https://www.kaggle.com/frlemarchand/bird-song-classification-using-an-efficientnet\n# Make sure to check out the entire notebook.\n\n# Load metadata file\ntrain = pd.read_csv('../input/birdclef-2021/train_metadata.csv',)\n\n# Limit the number of training samples and classes\n\n\n# Second, assume that birds with the most training samples are also the most common\n# A species needs at least 200 recordings with a rating above 4 to be considered common\nbirds_count = {}\nfor bird_species, count in zip(train.primary_label.unique(), \n                               train.groupby('primary_label')['primary_label'].count().values):\n    birds_count[bird_species] = count\nmost_represented_birds = [key for key,value in birds_count.items()] \n\nTRAIN = train.query('primary_label in @most_represented_birds')\nLABELS = sorted(TRAIN.primary_label.unique())\n\n# Let's see how many species and samples we have left\nprint('NUMBER OF SPECIES IN TRAIN DATA:', len(LABELS))\nprint('NUMBER OF SAMPLES IN TRAIN DATA:', len(TRAIN))\nprint('LABELS:', most_represented_birds)","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:37:44.726853Z","iopub.execute_input":"2021-06-08T09:37:44.727304Z","iopub.status.idle":"2021-06-08T09:37:45.242267Z","shell.execute_reply.started":"2021-06-08T09:37:44.727238Z","shell.execute_reply":"2021-06-08T09:37:45.24045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = np.load(\"../input/makingnumpy/train_label_0To170k.npy\")\ntrain_specs = np.load(\"../input/makingnumpy/train_specs_0To170k.npy\")","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:37:45.244727Z","iopub.execute_input":"2021-06-08T09:37:45.245192Z","iopub.status.idle":"2021-06-08T09:38:26.685525Z","shell.execute_reply.started":"2021-06-08T09:37:45.245151Z","shell.execute_reply":"2021-06-08T09:38:26.684493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_labels.shape)\nprint(train_specs.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:38:26.687415Z","iopub.execute_input":"2021-06-08T09:38:26.687806Z","iopub.status.idle":"2021-06-08T09:38:26.699145Z","shell.execute_reply.started":"2021-06-08T09:38:26.687741Z","shell.execute_reply":"2021-06-08T09:38:26.69793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_integer = np.array(train_labels)\ntrain_labels_integer = np.argmax(train_labels_integer, axis=1)\ntrain_labels_DF = pd.DataFrame(train_labels_integer, columns = [\"Col\"])\ntrain_labels_DF.value_counts()[:5]","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:38:26.700796Z","iopub.execute_input":"2021-06-08T09:38:26.701796Z","iopub.status.idle":"2021-06-08T09:38:26.90459Z","shell.execute_reply.started":"2021-06-08T09:38:26.701679Z","shell.execute_reply":"2021-06-08T09:38:26.903354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_integer","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:38:26.906513Z","iopub.execute_input":"2021-06-08T09:38:26.907028Z","iopub.status.idle":"2021-06-08T09:38:26.915723Z","shell.execute_reply.started":"2021-06-08T09:38:26.906988Z","shell.execute_reply":"2021-06-08T09:38:26.914359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_Classes_List = []\nfor i in range(0,397):\n    count = 0\n    list_current_class = []\n    for j in range(len(train_labels_integer)):\n        #print(j)\n        if train_labels_integer[j] == i:\n            count+=1\n            list_current_class.append(j)\n        if count > 500:\n            break\n            \n    all_Classes_List.append(list_current_class)\n    #print(i,\" Class == \",count)\n        \n\n    ","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:38:26.917621Z","iopub.execute_input":"2021-06-08T09:38:26.918403Z","iopub.status.idle":"2021-06-08T09:39:16.037134Z","shell.execute_reply.started":"2021-06-08T09:38:26.91836Z","shell.execute_reply":"2021-06-08T09:39:16.036016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_new = []\ntrain_specs_new = []\nfor i in all_Classes_List:\n    for j in i:\n        train_labels_new.append(train_labels[j])\n        train_specs_new.append(train_specs[j])\n\ntrain_labels = []\ntrain_specs = []","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:39:16.040273Z","iopub.execute_input":"2021-06-08T09:39:16.040724Z","iopub.status.idle":"2021-06-08T09:39:16.184176Z","shell.execute_reply.started":"2021-06-08T09:39:16.040682Z","shell.execute_reply":"2021-06-08T09:39:16.183041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_integer = np.array(train_labels_new)\ntrain_labels_integer = np.argmax(train_labels_integer, axis=1)\ntrain_labels_DF = pd.DataFrame(train_labels_integer, columns = [\"Col\"])\n#train_labels_DF.value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:39:16.186236Z","iopub.execute_input":"2021-06-08T09:39:16.186743Z","iopub.status.idle":"2021-06-08T09:39:16.404541Z","shell.execute_reply.started":"2021-06-08T09:39:16.1867Z","shell.execute_reply":"2021-06-08T09:39:16.40344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels_new = np.array(train_labels_new)\ntrain_specs_new = np.array(train_specs_new)\n\nprint(train_labels_new.shape,train_specs_new.shape)","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:39:16.406362Z","iopub.execute_input":"2021-06-08T09:39:16.406778Z","iopub.status.idle":"2021-06-08T09:39:18.242428Z","shell.execute_reply.started":"2021-06-08T09:39:16.406736Z","shell.execute_reply":"2021-06-08T09:39:18.241276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## melspec from others","metadata":{}},{"cell_type":"code","source":"MEL_PATHS = sorted(Path(\"../input\").glob(\"kkiller-birdclef-mels-computer-d7-part?/rich_train_metadata.csv\"))\nTRAIN_LABEL_PATHS = sorted(Path(\"../input\").glob(\"kkiller-birdclef-mels-computer-d7-part?/LABEL_IDS.json\"))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Simple Conv using Tensorflow","metadata":{}},{"cell_type":"code","source":"# Make sure your experiments are reproducible\ntf.random.set_seed(RANDOM_SEED)\n\n# Build a simple model as a sequence of  convolutional blocks.\n# Each block has the sequence CONV --> RELU --> BNORM --> MAXPOOL.\n# Finally, perform global average pooling and add 2 dense layers.\n# The last layer is our classification layer and is softmax activated.\n# (Well it's a multi-label task so sigmoid might actually be a better choice)\nmodel = tf.keras.Sequential([\n    \n    # First conv block\n    tf.keras.layers.Conv2D(16, (3, 3), activation='relu', \n                           input_shape=(SPEC_SHAPE[0], SPEC_SHAPE[1], 1)),\n    #tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    \n    #model.add(ResNet50(include_top = False, pooling = RESNET50_POOLING_AVERAGE,)\n    # Second conv block\n    tf.keras.layers.Conv2D(64, (3, 3), activation='relu'),\n    #tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n    # Third conv block\n    tf.keras.layers.Conv2D(128, (3, 3), activation='relu'),\n    #tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)), \n    \n    # Fourth conv block\n    tf.keras.layers.Conv2D(256, (3, 3), activation='relu'),\n    #tf.keras.layers.BatchNormalization(),\n    tf.keras.layers.MaxPooling2D((2, 2)),\n    \n    # Global pooling instead of flatten()\n    tf.keras.layers.GlobalAveragePooling2D(), \n    \n    # Dense block\n    tf.keras.layers.Dense(512, activation='relu'),   \n    tf.keras.layers.Dropout(0.5),  \n    tf.keras.layers.Dense(512, activation='relu'),   \n    tf.keras.layers.Dropout(0.5),\n    \n    # Classification layer\n    tf.keras.layers.Dense(len(LABELS), activation='softmax')\n])\nprint('MODEL HAS {} PARAMETERS.'.format(model.count_params()))","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:39:18.244179Z","iopub.execute_input":"2021-06-08T09:39:18.244871Z","iopub.status.idle":"2021-06-08T09:39:20.999692Z","shell.execute_reply.started":"2021-06-08T09:39:18.244794Z","shell.execute_reply":"2021-06-08T09:39:20.998372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=tf.keras.optimizers.Adam(lr=0.001),\n              loss=tf.keras.losses.CategoricalCrossentropy(label_smoothing=0.01),\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:39:21.001484Z","iopub.execute_input":"2021-06-08T09:39:21.00217Z","iopub.status.idle":"2021-06-08T09:39:21.022468Z","shell.execute_reply.started":"2021-06-08T09:39:21.002123Z","shell.execute_reply":"2021-06-08T09:39:21.021422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callbacks = [\n             tf.keras.callbacks.ModelCheckpoint(filepath='best_model.h5', \n                                                monitor='val_loss',\n                                                verbose=0,\n                                                save_best_only=True),\n            tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', patience=2, verbose=1, factor=0.5),\n            tf.keras.callbacks.EarlyStopping(monitor='val_loss',verbose=1,patience=5)]","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:39:21.023874Z","iopub.execute_input":"2021-06-08T09:39:21.024544Z","iopub.status.idle":"2021-06-08T09:39:21.03123Z","shell.execute_reply.started":"2021-06-08T09:39:21.0245Z","shell.execute_reply":"2021-06-08T09:39:21.029597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_specs_new, \n          train_labels_new,\n          batch_size=32,\n          validation_split=0.2,\n          callbacks=callbacks,\n          epochs=25)","metadata":{"execution":{"iopub.status.busy":"2021-06-08T09:39:21.033327Z","iopub.execute_input":"2021-06-08T09:39:21.033859Z","iopub.status.idle":"2021-06-08T09:47:16.57854Z","shell.execute_reply.started":"2021-06-08T09:39:21.033709Z","shell.execute_reply":"2021-06-08T09:47:16.576447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Trying Resnet","metadata":{}},{"cell_type":"code","source":"# Python ≥3.5 is required\nimport sys\nassert sys.version_info >= (3, 5)\n\n# Scikit-Learn ≥0.20 is required\nimport sklearn\nassert sklearn.__version__ >= \"0.20\"\n\n# Common imports\nimport numpy as np\nimport os\nimport gc\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.metrics import f1_score, recall_score, precision_score\n\n\nfrom keras import layers\nfrom keras.layers import Input, Add, Dense, Activation, ZeroPadding2D, BatchNormalization, Flatten, Conv2D, AveragePooling2D, MaxPooling2D, GlobalMaxPooling2D\nfrom keras.models import Model, load_model\nfrom keras.preprocessing import image\nfrom keras.utils import layer_utils, to_categorical\nfrom keras.utils.data_utils import get_file\nfrom keras.callbacks import Callback\nfrom keras.applications.imagenet_utils import preprocess_input\nimport pydot\nfrom IPython.display import SVG\nfrom keras.utils.vis_utils import model_to_dot\nfrom keras.utils import plot_model\nfrom keras.initializers import glorot_uniform\nimport scipy.misc\nfrom matplotlib.pyplot import imshow\nimport keras.backend as K\nK.set_image_data_format('channels_last')\nK.set_learning_phase(1)\nimport numpy as np\nimport pandas as pd\nimport os\nfrom sklearn.model_selection import KFold, StratifiedKFold\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n%matplotlib inline \n#plotting directly without requering the plot()\n\nimport warnings\nwarnings.filterwarnings(action=\"ignore\") #ignoring most of warnings, cleaning up the notebook for better visualization\n\npd.set_option('display.max_columns', 500) #fixing the number of rows and columns to be displayed\npd.set_option('display.max_rows', 500)\n\nprint(os.listdir(\"../input\")) #showing all the files in the ../input directory","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"k = 7\n\ntrain_labels_integer = np.array(train_labels)\ntrain_labels_integer = np.argmax(train_labels_integer, axis=1)\n\nfolds = list(StratifiedKFold(n_splits=k, shuffle=True, random_state=1).split(train_specs,train_labels_integer))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_labels.shape)\nprint(train_labels_integer.shape)\nprint(train_specs.shape)\nprint(len(folds[0][0])+len(folds[0][1]))\n\nprint(folds[0][0][3],folds[0][1][3])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''class Metrics(Callback):\n    def on_train_begin(self, logs={}):\n        self.val_f1s = []\n        self.val_recalls = []\n        self.val_precisions = []\n\n    def on_epoch_end(self, epoch, logs={}):\n        X_val, y_val = self.validation_data[:2]\n        y_pred = self.model.predict(X_val)\n\n        y_pred_cat = to_categorical(\n            y_pred.argmax(axis=1),\n            num_classes=14\n        )\n\n        _val_f1 = f1_score(y_val, y_pred_cat, average='macro')\n        _val_recall = recall_score(y_val, y_pred_cat, average='macro')\n        _val_precision = precision_score(y_val, y_pred_cat, average='macro')\n\n        self.val_f1s.append(_val_f1)\n        self.val_recalls.append(_val_recall)\n        self.val_precisions.append(_val_precision)\n\n        print((f\"val_f1: {_val_f1:.4f}\"\n               f\" — val_precision: {_val_precision:.4f}\"\n               f\" — val_recall: {_val_recall:.4f}\"))\n\n        return\n\nf1_metrics = Metrics()'''","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#identity_block\n\ndef identity_block(X, f, filters, stage, block):\n    \"\"\"\n    Implementation of the identity block as defined in Figure 3\n    \n    Arguments:\n    X -- input tensor of shape (m, n_H_prev, n_W_prev, n_C_prev)\n    f -- integer, specifying the shape of the middle CONV's window for the main path\n    filters -- python list of integers, defining the number of filters in the CONV layers of the main path\n    stage -- integer, used to name the layers, depending on their position in the network\n    block -- string/character, used to name the layers, depending on their position in the network\n    \n    Returns:\n    X -- output of the identity block, tensor of shape (n_H, n_W, n_C)\n    \"\"\"\n    \n    # defining name basis\n    conv_name_base = 'res' + str(stage) + block + '_branch'\n    bn_name_base = 'bn' + str(stage) + block + '_branch'\n    \n    # Retrieve Filters\n    F1, F2, F3 = filters\n    \n    # Save the input value. You'll need this later to add back to the main path. \n    X_shortcut = X\n    \n    # First component of main path\n    X = Conv2D(filters = F1, kernel_size = (1, 1), strides = (1,1), padding = 'valid', name = conv_name_base + '2a', kernel_initializer = glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis = 3, name = bn_name_base + '2a')(X)\n    X = Activation('relu')(X)\n    \n    # Second component of main path\n    X = Conv2D(filters = F2, kernel_size = (f, f), strides = (1,1), padding = 'same', name = conv_name_base + '2b', kernel_initializer = glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis = 3, name = bn_name_base + '2b')(X)\n    X = Activation('relu')(X)\n\n    # Third component of main path\n    X = Conv2D(filters = F3, kernel_size = (1, 1), strides = (1,1), padding = 'valid', name = conv_name_base + '2c', kernel_initializer = glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis = 3, name = bn_name_base + '2c')(X)\n\n    # Final step: Add shortcut value to main path, and pass it through a RELU activation\n    X = Add()([X,X_shortcut])\n    X = Activation('relu')(X)\n    \n    return X\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def convolutional_block(X, f, filters, stage, block, s = 2):\n    \"\"\"\n    Implementation of the convolutional block as defined in Figure 4\n    \n    Arguments:\n    X -- input tensor of shape (m, n_H_prev, n_W_prev, n_C_prev)\n    f -- integer, specifying the shape of the middle CONV's window for the main path\n    filters -- python list of integers, defining the number of filters in the CONV layers of the main path\n    stage -- integer, used to name the layers, depending on their position in the network\n    block -- string/character, used to name the layers, depending on their position in the network\n    s -- Integer, specifying the stride to be used\n    \n    Returns:\n    X -- output of the convolutional block, tensor of shape (n_H, n_W, n_C)\n    \"\"\"\n    \n    # defining name basis\n    conv_name_base = 'res' + str(stage) + block + '_branch'\n    bn_name_base = 'bn' + str(stage) + block + '_branch'\n    \n    # Retrieve Filters\n    F1, F2, F3 = filters\n    \n    # Save the input value\n    X_shortcut = X\n\n\n    ##### MAIN PATH #####\n    # First component of main path \n    X = Conv2D(F1, (1, 1), strides = (s,s), name = conv_name_base + '2a',padding = 'valid', kernel_initializer = glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis = 3, name = bn_name_base + '2a')(X)\n    X = Activation('relu')(X)\n\n    # Second component of main path\n    X = Conv2D(F2, (f, f), strides = (1,1), name = conv_name_base + '2b', padding = 'same', kernel_initializer = glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis = 3, name = bn_name_base + '2b')(X)\n    X = Activation('relu')(X)\n\n    # Third component of main path\n    X = Conv2D(F3, (1, 1), strides = (1,1), name = conv_name_base + '2c', padding = 'valid', kernel_initializer = glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis = 3, name = bn_name_base + '2c')(X)\n\n    ##### SHORTCUT PATH ####\n    X_shortcut = Conv2D(F3, (1, 1), strides = (s,s), name = conv_name_base + '1', padding = 'valid', kernel_initializer = glorot_uniform(seed=0))(X_shortcut)\n    X_shortcut = BatchNormalization(axis = 3, name = bn_name_base + '1')(X_shortcut)\n\n    # Final step: Add shortcut value to main path, and pass it through a RELU activation\n    X = Add()([X,X_shortcut])\n    X = Activation('relu')(X)\n    \n    return X","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ResNet50(input_shape = (32, 32, 3), classes = 14):\n    \"\"\"\n    Implementation of the popular ResNet50 the following architecture:\n    CONV2D -> BATCHNORM -> RELU -> MAXPOOL -> CONVBLOCK -> IDBLOCK*2 -> CONVBLOCK -> IDBLOCK*3\n    -> CONVBLOCK -> IDBLOCK*5 -> CONVBLOCK -> IDBLOCK*2 -> AVGPOOL -> TOPLAYER\n\n    Arguments:\n    input_shape -- shape of the images of the dataset\n    classes -- integer, number of classes\n\n    Returns:\n    model -- a Model() instance in Keras\n    \"\"\"\n    \n    # Define the input as a tensor with shape input_shape\n    X_input = Input(input_shape)\n\n    \n    # Zero-Padding\n    X = ZeroPadding2D((3, 3))(X_input)\n    \n    # Stage 1\n    X = Conv2D(32, (7, 7), strides = (1, 1), name = 'conv1', kernel_initializer = glorot_uniform(seed=0))(X)\n    X = BatchNormalization(axis = 3, name = 'bn_conv1')(X)\n    X = Activation('relu')(X)\n    X = MaxPooling2D((3, 3))(X)\n\n    # Stage 2\n    X = convolutional_block(X, f = 3, filters = [32, 32, 128], stage = 2, block='a', s = 1)\n    X = identity_block(X, 3, [32, 32, 128], stage=2, block='b')\n    X = identity_block(X, 3, [32, 32, 128], stage=2, block='c')\n\n    # Stage 3\n    X = convolutional_block(X, f = 3, filters = [64, 64, 256], stage = 3, block='a', s = 2)\n    X = identity_block(X, 3, [64, 64, 256], stage=3, block='b')\n    X = identity_block(X, 3, [64, 64, 256], stage=3, block='c')\n    X = identity_block(X, 3, [64, 64, 256], stage=3, block='d')\n\n    # Stage 4 \n    X = convolutional_block(X, f = 3, filters = [128, 128, 512], stage = 4, block='a', s = 2)\n    X = identity_block(X, 3, [128, 128, 512], stage=4, block='b')\n    X = identity_block(X, 3, [128, 128, 512], stage=4, block='c')\n    X = identity_block(X, 3, [128, 128, 512], stage=4, block='d')\n    X = identity_block(X, 3, [128, 128, 512], stage=4, block='e')\n    X = identity_block(X, 3, [128, 128, 512], stage=4, block='f')\n\n    # Stage 5 \n    X = convolutional_block(X, f = 3, filters = [256,256, 1024], stage = 5, block='a', s = 2)\n    X = identity_block(X, 3, [256,256, 1024], stage=5, block='b')\n    X = identity_block(X, 3, [256,256, 1024], stage=5, block='c')\n\n    # AVGPOOL\n    X = AveragePooling2D(pool_size=(2,2), name='avg_pool')(X)\n    \n\n    # output layer\n    X = Flatten()(X)\n    X = Dense(classes, activation='softmax', name='fc' + str(classes), kernel_initializer = glorot_uniform(seed=0))(X)\n    \n    \n    # Create model\n    model = Model(inputs = X_input, outputs = X, name='ResNet50')\n\n    return model","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = ResNet50(input_shape = (SPEC_SHAPE[0], SPEC_SHAPE[1], 1), classes = len(LABELS))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.callbacks import ReduceLROnPlateau\nfrom keras.callbacks import EarlyStopping\n\nreduce_lr = ReduceLROnPlateau(monitor='val_acc', factor=0.2,patience=5, min_lr=1e-5)\nes = EarlyStopping(monitor='val_loss', mode='min', verbose=1, patience=5)\ncallbacks = [reduce_lr,es]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam', loss='categorical_crossentropy', metrics=['accuracy'])\n#model.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for j, (train_idx, val_idx) in enumerate(folds):\n    \n    print('\\nFold ',j)\n    try:\n        train_specs = train_specs[train_idx]\n        train_labels = train_labels[train_idx]\n    \n        train_specs_val = train_specs[val_idx]\n        train_labels_val = train_labels[val_idx]\n    except:\n        print(\"Error\")\n    name_weights = \"final_model_fold\" + str(j) + \"_weights.h5\"\n    reduce_lr = ReduceLROnPlateau(monitor='val_acc', factor=0.2,patience=5, min_lr=1e-5)\n    es = EarlyStopping(monitor='val_loss', mode='min', verbose=1, patience=5)\n    callbacks = [reduce_lr,es]\n    #generator = gen.flow(X_train_cv, y_train_cv, batch_size = batch_size)\n    model = ResNet50(input_shape = (SPEC_SHAPE[0], SPEC_SHAPE[1], 1), classes = len(LABELS))\n    model.fit(\n                train_specs,\n                train_labels,\n                steps_per_epoch=len(train_specs)/16,\n                epochs=10,\n                shuffle=True,\n                verbose=1,\n                validation_data = (train_specs_val, train_labels_val),\n                callbacks = callbacks)\n    \n    print(model.evaluate(X_valid_cv, y_valid_cv))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_specs,\n    train_labels,          \n    batch_size=16,\n    epochs=10,\n    validation_split=0.4,\n    callbacks = callbacks\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.subplots(figsize=(12,10))\nplt.plot(history.history['loss'], color='b', label=\"Training loss\")\nplt.plot(history.history['val_loss'], color='r', label=\"validation loss\")\nplt.legend(loc='best', shadow=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test","metadata":{}},{"cell_type":"code","source":"# Load the best checkpoint\nmodel = tf.keras.models.load_model('best_model.h5')\n\n# Pick a soundscape\nsoundscape_path = '../input/birdclef-2021/train_soundscapes/28933_SSW_20170408.ogg'\n\n# Open it with librosa\nsig, rate = librosa.load(soundscape_path, sr=SAMPLE_RATE)\n\n# Store results so that we can analyze them later\ndata = {'row_id': [], 'prediction': [], 'score': []}\n\n# Split signal into 5-second chunks\n# Just like we did before (well, this could actually be a seperate function)\nsig_splits = []\nfor i in range(0, len(sig), int(SIGNAL_LENGTH * SAMPLE_RATE)):\n    split = sig[i:i + int(SIGNAL_LENGTH * SAMPLE_RATE)]\n\n    # End of signal?\n    if len(split) < int(SIGNAL_LENGTH * SAMPLE_RATE):\n        break\n\n    sig_splits.append(split)\n    \n# Get the spectrograms and run inference on each of them\n# This should be the exact same process as we used to\n# generate training samples!\nseconds, scnt = 0, 0\nfor chunk in sig_splits:\n    \n    # Keep track of the end time of each chunk\n    seconds += 5\n        \n    # Get the spectrogram\n    hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n    mel_spec = librosa.feature.melspectrogram(y=chunk, \n                                              sr=SAMPLE_RATE, \n                                              n_fft=1024, \n                                              hop_length=hop_length, \n                                              n_mels=SPEC_SHAPE[0], \n                                              fmin=FMIN, \n                                              fmax=FMAX)\n\n    mel_spec = librosa.power_to_db(mel_spec, ref=np.max) \n\n    # Normalize to match the value range we used during training.\n    # That's something you should always double check!\n    mel_spec -= mel_spec.min()\n    mel_spec /= mel_spec.max()\n    \n    # Add channel axis to 2D array\n    mel_spec = np.expand_dims(mel_spec, -1)\n\n    # Add new dimension for batch size\n    mel_spec = np.expand_dims(mel_spec, 0)\n    \n    # Predict\n    p = model.predict(mel_spec)[0]\n    \n    # Get highest scoring species\n    idx = p.argmax()\n    species = LABELS[idx]\n    score = p[idx]\n    \n    # Prepare submission entry\n    data['row_id'].append(soundscape_path.split(os.sep)[-1].rsplit('_', 1)[0] + \n                          '_' + str(seconds))    \n    \n    # Decide if it's a \"nocall\" or a species by applying a threshold\n    if score > 0.25:\n        data['prediction'].append(species)\n        scnt += 1\n    else:\n        data['prediction'].append('nocall')\n        \n    # Add the confidence score as well\n    data['score'].append(score)\n        \nprint('SOUNSCAPE ANALYSIS DONE. FOUND {} BIRDS.'.format(scnt))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = pd.DataFrame(data, columns = ['row_id', 'prediction', 'score'])\n\n# Merge with ground truth so we can inspect\ngt = pd.read_csv('../input/birdclef-2021/train_soundscape_labels.csv',)\nresults = pd.merge(gt, results, on='row_id')\n\n# Let's look at the first 50 entries\nresults.head(50)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}