{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8033468,"sourceType":"datasetVersion","datasetId":4735360},{"sourceId":8462031,"sourceType":"datasetVersion","datasetId":4773478}],"dockerImageVersionId":30683,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \nimport os\nimport glob\nimport shutil\n\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom tensorflow.keras import layers, models\n\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.applications import NASNetMobile , VGG19\nfrom tensorflow.keras.layers import Input, Dense, GlobalAveragePooling2D, concatenate, LSTM, Bidirectional, Reshape, Dropout\nfrom tensorflow.keras.models import Model\n\nfrom sklearn.preprocessing import LabelBinarizer\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import roc_auc_score, roc_curve, auc\n\nimport time\n\nfrom PIL import Image, ImageOps\n\nfrom tensorflow.keras.preprocessing import image\n\nimport librosa\nimport librosa.display\nfrom tensorflow.keras.models import load_model\n\nimport matplotlib.cm as cm\n\nAUTOTUNE = tf.data.experimental.AUTOTUNE","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-23T07:36:47.644177Z","iopub.execute_input":"2024-05-23T07:36:47.644511Z","iopub.status.idle":"2024-05-23T07:37:00.212028Z","shell.execute_reply.started":"2024-05-23T07:36:47.644482Z","shell.execute_reply":"2024-05-23T07:37:00.211141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data preprocesing","metadata":{}},{"cell_type":"code","source":"# Base directory where the folders are stored\nimage_folder = \"/kaggle/input/birdclef-2024-mel-spectrograms/train_images/\"\n\n# Mapping of classes to numerical labels\n# Classes are alphabetically sorted - so we can easily restore class order when we load\nclass_labels = {class_name: i for i, class_name in enumerate(sorted(os.listdir(image_folder)))}\nnum_classes = len(class_labels)\n\n# Collect all file paths and their corresponding class labels\nfile_paths = []\nlabels = []\n\n# New structure to keep track of groups\nsamples = {}\n\nfor class_name in os.listdir(image_folder):\n    class_dir = os.path.join(image_folder, class_name)\n    for filename in os.listdir(class_dir):\n        # Extract base sample name from filename (files split at \"_\")\n        sample_base = filename.split('_')[0] \n        full_path = os.path.join(class_dir, filename)\n        \n        if sample_base not in samples:\n            samples[sample_base] = {'files': [], 'label': class_labels[class_name]}\n        samples[sample_base]['files'].append(full_path)","metadata":{"execution":{"iopub.status.busy":"2024-05-23T07:37:05.599867Z","iopub.execute_input":"2024-05-23T07:37:05.600464Z","iopub.status.idle":"2024-05-23T07:37:11.621950Z","shell.execute_reply.started":"2024-05-23T07:37:05.600433Z","shell.execute_reply":"2024-05-23T07:37:11.621165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_image(file_path, label):\n    img = tf.io.read_file(file_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.random_contrast(img, lower=0.2, upper=1.8)\n    img = tf.image.random_brightness(img, max_delta=0.2)\n\n    #scale 0-1\n    img = tf.cast(img, tf.float32) / 255.0\n    \n    label = tf.one_hot(label, depth=num_classes)\n    return img, label","metadata":{"execution":{"iopub.status.busy":"2024-05-23T07:37:19.350439Z","iopub.execute_input":"2024-05-23T07:37:19.350794Z","iopub.status.idle":"2024-05-23T07:37:19.356990Z","shell.execute_reply.started":"2024-05-23T07:37:19.350770Z","shell.execute_reply":"2024-05-23T07:37:19.355999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_size = 0.2\nbatch_size = 32\n\n\nsamples_list = list(samples.items())\n\n\ntrain_samples, val_samples = train_test_split(samples_list, test_size=test_size, random_state=42)\n\n\ndef extract_files_and_labels(sample_list):\n    file_paths = []\n    labels = []\n    for _, sample_info in sample_list:\n        file_paths.extend(sample_info['files'])\n        labels.extend([sample_info['label']] * len(sample_info['files']))\n    return file_paths, labels\n\ntrain_files, train_labels = extract_files_and_labels(train_samples)\nval_files, val_labels = extract_files_and_labels(val_samples)\n\nnum_classes = np.max(train_labels) + 1\n\n#tf.data.Dataset\ntrain_dataset = tf.data.Dataset.from_tensor_slices((train_files, train_labels))\ntrain_dataset = train_dataset.map(preprocess_image, num_parallel_calls=AUTOTUNE)\ntrain_dataset = train_dataset.shuffle(buffer_size=1000).batch(batch_size).prefetch(AUTOTUNE)\n\nval_dataset = tf.data.Dataset.from_tensor_slices((val_files, val_labels))\nval_dataset = val_dataset.map(preprocess_image, num_parallel_calls=AUTOTUNE)\nval_dataset = val_dataset.batch(batch_size).prefetch(AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2024-05-23T07:37:25.673966Z","iopub.execute_input":"2024-05-23T07:37:25.674324Z","iopub.status.idle":"2024-05-23T07:37:26.650622Z","shell.execute_reply.started":"2024-05-23T07:37:25.674285Z","shell.execute_reply":"2024-05-23T07:37:26.649660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CNN (MobileNetV2 on ImageNet)","metadata":{}},{"cell_type":"code","source":"#MobileNetV2 ваги з imagenet\nfrom tensorflow.keras.applications import MobileNetV2\nbase_model = MobileNetV2(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n\nbase_model.trainable = True\n\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\n\n# FC слой\nx = Dense(1024, activation='relu')(x)\n\n# вихід слоя -> клас\npredictions = Dense(num_classes, activation='softmax')(x)\n\n# Lr\ninitial_learning_rate = 0.00005\nlr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n    initial_learning_rate=initial_learning_rate,\n    decay_steps=1000,\n    decay_rate=0.96,\n    staircase=True)\n\noptimizer = tf.keras.optimizers.Adam(learning_rate=lr_schedule)\n\n# This is the model we will train\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\nmodel.compile(optimizer=optimizer,\n              loss='categorical_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-05-22T20:15:26.813366Z","iopub.execute_input":"2024-05-22T20:15:26.814046Z","iopub.status.idle":"2024-05-22T20:15:28.121253Z","shell.execute_reply.started":"2024-05-22T20:15:26.814013Z","shell.execute_reply":"2024-05-22T20:15:28.120466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_dataset,\n    epochs=15,\n    validation_data=val_dataset\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-22T18:38:03.531468Z","iopub.execute_input":"2024-05-22T18:38:03.532364Z","iopub.status.idle":"2024-05-22T18:50:19.562019Z","shell.execute_reply.started":"2024-05-22T18:38:03.532332Z","shell.execute_reply":"2024-05-22T18:50:19.561121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_image_pred(pil_img):\n    img = pil_img.convert('RGB')\n    img_array = image.img_to_array(img)\n    img_array = np.expand_dims(img_array, axis=0)\n    img_array /= 255.0\n    return img_array\n\ndef make_prediction(image):\n    img_array = preprocess_image_pred(image)\n    predictions = model.predict(img_array)\n        \n    return predictions\n\npil_image = Image.open(\"/kaggle/input/birdclef-2024-mel-spectrograms/train_images/asbfly/XC164848_00.png\")\ndisplay(pil_image)\n\npredictions = make_prediction(pil_image)\nprint(\"Asbfly: \", predictions[0][0])\nprint(\"\\nAll preds:\\n\", predictions[0])","metadata":{"execution":{"iopub.status.busy":"2024-05-22T18:55:05.131994Z","iopub.execute_input":"2024-05-22T18:55:05.132436Z","iopub.status.idle":"2024-05-22T18:55:10.037061Z","shell.execute_reply.started":"2024-05-22T18:55:05.132403Z","shell.execute_reply":"2024-05-22T18:55:10.036088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('birdclef2024_imagenet.keras')","metadata":{"execution":{"iopub.status.busy":"2024-05-22T18:55:16.901940Z","iopub.execute_input":"2024-05-22T18:55:16.902666Z","iopub.status.idle":"2024-05-22T18:55:17.556033Z","shell.execute_reply.started":"2024-05-22T18:55:16.902634Z","shell.execute_reply":"2024-05-22T18:55:17.554993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"original_model_path = '/kaggle/input/birdclef24-spectr-imagenettrained-model/birdclef2024_imagenet.keras'\ntemp_model_path = '/kaggle/working/birdclef2024_imagenet.keras'\n\n\nshutil.copyfile(original_model_path, temp_model_path)","metadata":{"execution":{"iopub.status.busy":"2024-05-22T20:17:45.303074Z","iopub.execute_input":"2024-05-22T20:17:45.303449Z","iopub.status.idle":"2024-05-22T20:17:46.064057Z","shell.execute_reply.started":"2024-05-22T20:17:45.303419Z","shell.execute_reply":"2024-05-22T20:17:46.063040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#new_model = tf.keras.models.load_model('/kaggle/working/birdclef2024_imagenet.keras')","metadata":{"execution":{"iopub.status.busy":"2024-05-22T20:17:49.241007Z","iopub.execute_input":"2024-05-22T20:17:49.241385Z","iopub.status.idle":"2024-05-22T20:17:52.160768Z","shell.execute_reply.started":"2024-05-22T20:17:49.241356Z","shell.execute_reply":"2024-05-22T20:17:52.159880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get predictions on validation data\nval_predictions = new_model.predict(val_dataset)\n\n# Convert tensors to numpy arrays\nval_labels_np = np.concatenate([y.numpy() for x, y in val_dataset], axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-05-22T20:19:30.383795Z","iopub.execute_input":"2024-05-22T20:19:30.384634Z","iopub.status.idle":"2024-05-22T20:19:55.995578Z","shell.execute_reply.started":"2024-05-22T20:19:30.384599Z","shell.execute_reply":"2024-05-22T20:19:55.994758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\n\n# Assuming val_labels_np and val_predictions are your validation labels and predictions respectively\n# Convert the predicted probabilities to class labels by taking the argmax along axis 1\npredicted_classes = val_predictions.argmax(axis=1)\n\n# Convert one-hot encoded labels to class labels by taking the argmax along axis 1\ntrue_classes = val_labels_np.argmax(axis=1)\n\n# Compute accuracy\naccuracy = accuracy_score(true_classes, predicted_classes)\nprint(\"Accuracy:\", accuracy)\n\n# Compute precision\nprecision = precision_score(true_classes, predicted_classes, average='weighted')\nprint(\"Precision:\", precision)\n\n# Compute recall\nrecall = recall_score(true_classes, predicted_classes, average='weighted')\nprint(\"Recall:\", recall)\n\n# Compute F1-score\nf1 = f1_score(true_classes, predicted_classes, average='weighted')\nprint(\"F1-score:\", f1)","metadata":{"execution":{"iopub.status.busy":"2024-05-22T20:20:38.950599Z","iopub.execute_input":"2024-05-22T20:20:38.951084Z","iopub.status.idle":"2024-05-22T20:20:38.987466Z","shell.execute_reply.started":"2024-05-22T20:20:38.951050Z","shell.execute_reply":"2024-05-22T20:20:38.986237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import roc_curve, auc\nfrom sklearn.preprocessing import label_binarize\n\n# Assuming val_labels_np and val_predictions are your validation labels and predictions respectively\n# Convert the one-hot encoded labels to binary format\ntrue_labels = np.argmax(val_labels_np, axis=1)\n# Binarize the true labels\ntrue_labels_binarized = label_binarize(true_labels, classes=np.arange(182))\n\n# Compute the ROC curve and ROC area for each class\nfpr = dict()\ntpr = dict()\nroc_auc = dict()\nfor i in range(182):\n    fpr[i], tpr[i], _ = roc_curve(true_labels_binarized[:, i], val_predictions[:, i])\n    if np.sum(true_labels_binarized[:, i]) > 0:  # Check if there are positive samples for the class\n        roc_auc[i] = auc(fpr[i], tpr[i])\n\n# Plot ROC curve for each class\nplt.figure(figsize=(8, 6))\nfor i in range(182):\n    if i in roc_auc:  # Plot only if ROC AUC score is available\n        plt.plot(fpr[i], tpr[i], label=f'Class {i} (AUC = {roc_auc[i]:.2f})')\n\nplt.plot([0, 1], [0, 1], 'k--', label='Random')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve for Multiclass Classification (One-vs-Rest)')\nplt.legend(loc='lower right')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-22T20:21:04.579421Z","iopub.execute_input":"2024-05-22T20:21:04.580175Z","iopub.status.idle":"2024-05-22T20:21:06.905599Z","shell.execute_reply.started":"2024-05-22T20:21:04.580125Z","shell.execute_reply":"2024-05-22T20:21:06.904580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# mean ROC/AUC\nmean_roc_auc = np.mean(list(roc_auc.values()))\nprint(\"Mean ROC/AUC Score:\", mean_roc_auc)","metadata":{"execution":{"iopub.status.busy":"2024-05-22T20:22:12.728007Z","iopub.execute_input":"2024-05-22T20:22:12.728432Z","iopub.status.idle":"2024-05-22T20:22:12.736707Z","shell.execute_reply.started":"2024-05-22T20:22:12.728399Z","shell.execute_reply":"2024-05-22T20:22:12.735717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CRNN (VGG19 on ImageNet and Recurrent layers)","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models\n\n\ndef create_crnn_model(input_shape, num_classes):\n    # Pre-trained VGG-19 model without FC\n    vgg_base = VGG19(weights='imagenet', include_top=False, input_shape=input_shape)\n    \n    # Freeze convolutional layers\n    for layer in vgg_base.layers:\n        layer.trainable = False\n    \n    # Extract features\n    vgg_output = vgg_base.output\n    \n    # Average Pooling\n    avg_pooling = layers.GlobalAveragePooling2D()(vgg_output)\n    # Convert features to sequence\n    reshaped_output = layers.Reshape((-1, 512))(vgg_output)\n    \n    # Recurrent layers\n    lstm_layer1 = layers.LSTM(64, return_sequences=True)(reshaped_output)\n    lstm_layer2 = layers.LSTM(64)(lstm_layer1)\n\n    # Dense layers\n    dense_layer1 = layers.Dense(512, activation='relu')(lstm_layer2)\n    output_layer = layers.Dense(num_classes, activation='softmax')(dense_layer1)\n    \n    # Create model\n    model = models.Model(inputs=vgg_base.input, outputs=output_layer)\n    \n    return model\n\n\ninput_shape = (224, 224, 3)  \nnum_classes = 182  \n\n# CRNN\nmodel = create_crnn_model(input_shape, num_classes)\n\n\nmodel.compile(optimizer='adam',\n              loss='categorical_crossentropy',\n              metrics=['accuracy'])\n\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-05-23T07:37:37.698108Z","iopub.execute_input":"2024-05-23T07:37:37.698500Z","iopub.status.idle":"2024-05-23T07:37:39.108009Z","shell.execute_reply.started":"2024-05-23T07:37:37.698470Z","shell.execute_reply":"2024-05-23T07:37:39.107159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.callbacks import LearningRateScheduler\n\n\ndef scheduler(epoch, lr):\n    if epoch < 10:\n        return lr\n    else:\n        return float( lr * tf.math.exp(-0.1))\n\n# Create the LearningRateScheduler callback\nlr_scheduler = LearningRateScheduler(scheduler)","metadata":{"execution":{"iopub.status.busy":"2024-05-23T07:37:50.252139Z","iopub.execute_input":"2024-05-23T07:37:50.252975Z","iopub.status.idle":"2024-05-23T07:37:50.258032Z","shell.execute_reply.started":"2024-05-23T07:37:50.252942Z","shell.execute_reply":"2024-05-23T07:37:50.257120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.get_logger().setLevel('ERROR')\n\n# Fit\nhistory = model.fit(train_dataset, \n                                     epochs=30,\n                                     batch_size=1,\n                                     validation_data=val_dataset,\n                                     callbacks=[lr_scheduler]\n                                    )\n\n\n# Evaluate\ntest_loss, test_acc = model.evaluate(val_dataset)\nprint('Test accuracy:', test_acc)","metadata":{"execution":{"iopub.status.busy":"2024-05-23T07:37:54.022524Z","iopub.execute_input":"2024-05-23T07:37:54.023427Z","iopub.status.idle":"2024-05-23T08:24:51.121570Z","shell.execute_reply.started":"2024-05-23T07:37:54.023394Z","shell.execute_reply":"2024-05-23T08:24:51.120584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('birdclef2024_rcnn.keras')","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:35:55.528918Z","iopub.execute_input":"2024-05-23T08:35:55.529691Z","iopub.status.idle":"2024-05-23T08:35:55.897661Z","shell.execute_reply.started":"2024-05-23T08:35:55.529655Z","shell.execute_reply":"2024-05-23T08:35:55.896617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for images, labels in val_dataset.take(1):\n    print(images.shape)","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:36:11.153800Z","iopub.execute_input":"2024-05-23T08:36:11.154455Z","iopub.status.idle":"2024-05-23T08:36:11.261332Z","shell.execute_reply.started":"2024-05-23T08:36:11.154422Z","shell.execute_reply":"2024-05-23T08:36:11.260375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get predictions on validation data\nval_predictions = model.predict(val_dataset)\n\n# Convert tensors to numpy arrays\nval_labels_np = np.concatenate([y.numpy() for x, y in val_dataset], axis=0)","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:36:13.382604Z","iopub.execute_input":"2024-05-23T08:36:13.383314Z","iopub.status.idle":"2024-05-23T08:36:35.169250Z","shell.execute_reply.started":"2024-05-23T08:36:13.383272Z","shell.execute_reply":"2024-05-23T08:36:35.168422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_labels_np , val_predictions","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:36:48.110324Z","iopub.execute_input":"2024-05-23T08:36:48.111206Z","iopub.status.idle":"2024-05-23T08:36:48.118856Z","shell.execute_reply.started":"2024-05-23T08:36:48.111172Z","shell.execute_reply":"2024-05-23T08:36:48.117840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score\n\n\npredicted_classes = val_predictions.argmax(axis=1)\n\n\ntrue_classes = val_labels_np.argmax(axis=1)\n\n# Compute accuracy\naccuracy = accuracy_score(true_classes, predicted_classes)\nprint(\"Accuracy:\", accuracy)\n\n# Compute precision\nprecision = precision_score(true_classes, predicted_classes, average='weighted')\nprint(\"Precision:\", precision)\n\n# Compute recall\nrecall = recall_score(true_classes, predicted_classes, average='weighted')\nprint(\"Recall:\", recall)\n\n# Compute F1-score\nf1 = f1_score(true_classes, predicted_classes, average='weighted')\nprint(\"F1-score:\", f1)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:36:58.410409Z","iopub.execute_input":"2024-05-23T08:36:58.410778Z","iopub.status.idle":"2024-05-23T08:36:58.436748Z","shell.execute_reply.started":"2024-05-23T08:36:58.410751Z","shell.execute_reply":"2024-05-23T08:36:58.435644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import roc_curve, auc\nfrom sklearn.preprocessing import label_binarize\n\n\ntrue_labels = np.argmax(val_labels_np, axis=1)\n# Binarize\ntrue_labels_binarized = label_binarize(true_labels, classes=np.arange(182))\n\n# ROC\nfpr = dict()\ntpr = dict()\nroc_auc = dict()\nfor i in range(182):\n    fpr[i], tpr[i], _ = roc_curve(true_labels_binarized[:, i], val_predictions[:, i])\n    if np.sum(true_labels_binarized[:, i]) > 0:  # Check if there are positive samples for the class\n        roc_auc[i] = auc(fpr[i], tpr[i])\n\n# Plot ROC \nplt.figure(figsize=(8, 6))\nfor i in range(182):\n    if i in roc_auc:  # Plot only if ROC AUC score is available\n        plt.plot(fpr[i], tpr[i], label=f'Class {i} (AUC = {roc_auc[i]:.2f})')\n\nplt.plot([0, 1], [0, 1], 'k--', label='Random')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve for Multiclass Classification (One-vs-Rest)')\nplt.legend(loc='lower right')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:37:23.045120Z","iopub.execute_input":"2024-05-23T08:37:23.045830Z","iopub.status.idle":"2024-05-23T08:37:25.575482Z","shell.execute_reply.started":"2024-05-23T08:37:23.045801Z","shell.execute_reply":"2024-05-23T08:37:25.574598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Mean ROC/AUC score\nmean_roc_auc = np.mean(list(roc_auc.values()))\nprint(\"Mean ROC/AUC Score:\", mean_roc_auc)","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:37:48.787762Z","iopub.execute_input":"2024-05-23T08:37:48.788136Z","iopub.status.idle":"2024-05-23T08:37:48.794379Z","shell.execute_reply.started":"2024-05-23T08:37:48.788108Z","shell.execute_reply":"2024-05-23T08:37:48.793252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submit","metadata":{}},{"cell_type":"code","source":"soundscapes_folder = \"/kaggle/input/birdclef-2024/test_soundscapes\"\nquick_test = False\n\n#if we don't have any files in test_soundscapes - revert to test mode\n\nif len(glob.glob(f\"{soundscapes_folder}/*.ogg\")) == 0:\n    soundscapes_folder = \"/kaggle/input/birdclef-2024/unlabeled_soundscapes\"\n    quick_test = True\n\n#spectrogram length\naudio_duration = 5\n\n#dimension of spectrograms\nimage_size = 224\n\n#we make the same predictions for this number of time indexes (for performance)\n#ie - value of 3 means making one prediction per 15 seconds - and then duplicating for the rest\n#1 = no skipping\nduplicate_predictions_count = 4","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:53:16.148081Z","iopub.execute_input":"2024-05-23T08:53:16.148826Z","iopub.status.idle":"2024-05-23T08:53:16.154995Z","shell.execute_reply.started":"2024-05-23T08:53:16.148791Z","shell.execute_reply":"2024-05-23T08:53:16.154094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_spectrograms_for_ogg(dirname, filename):\n\n    images = []\n    \n    # Load the entire audio file once\n    audio_path = os.path.join(dirname, filename)\n    audio_data, sr = librosa.load(audio_path, sr=None)\n    total_duration = librosa.get_duration(y=audio_data, sr=sr)\n    \n    num_segments = int(total_duration // audio_duration)\n\n    for segment in range(0, num_segments, duplicate_predictions_count):\n        offset_samples = int(segment * audio_duration * sr)\n        end_samples = int(offset_samples + audio_duration * sr)\n        \n        segment_data = audio_data[offset_samples:end_samples]\n        \n        # Generate the spectrogram\n\n        S = librosa.feature.melspectrogram(y=segment_data, sr=sr, n_mels=128)\n        S_db = librosa.amplitude_to_db(S, ref=np.max)\n                \n        #convert spectrogram data into directly into image (much faster than matplotlib)\n        normalized_array = (S_db - np.min(S_db)) / (np.max(S_db) - np.min(S_db))\n        \n        #set color mapping (so consistent with train)\n        spectrogram_image = cm.magma(normalized_array)[:, :, :3]\n        spectrogram_image = (spectrogram_image * 255).astype(np.uint8)\n        \n        spectrogram_image = Image.fromarray(spectrogram_image)\n        \n        #resize and flip (so consistent with train)\n        spectrogram_image = spectrogram_image.resize((image_size, image_size), Image.ANTIALIAS)\n        spectrogram_image = ImageOps.flip(spectrogram_image)\n\n        \n        images.append(spectrogram_image)\n\n    return images","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:53:30.710576Z","iopub.execute_input":"2024-05-23T08:53:30.710932Z","iopub.status.idle":"2024-05-23T08:53:30.722838Z","shell.execute_reply.started":"2024-05-23T08:53:30.710904Z","shell.execute_reply":"2024-05-23T08:53:30.721954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess_image(pil_img):\n    # Convert the image to RGB mode in case it's not\n    img = pil_img.convert('RGB')\n        \n    # Convert the PIL image to a numpy array\n    img_array = image.img_to_array(img)\n    \n    # Expand dimensions to have shape (1, 224, 224, 3)\n    img_array = np.expand_dims(img_array, axis=0)\n    \n    # Scale pixel values to [0, 1]\n    img_array /= 255.0\n    \n    return img_array\n\ndef make_prediction(image):\n    #classes were created in alphabetical order so can be loaded into DF with alpha column order\n    img_array = preprocess_image(image)\n    predictions = model.predict(img_array, verbose=None)    \n    return predictions\n\n\ndef make_prediction_batch(images):\n    # Preprocess each image and stack them into a single batch tensor\n    img_batch = np.stack([preprocess_image(image) for image in images])\n        \n    # Predict on the batch\n    predictions = model.predict(np.squeeze(img_batch), verbose=None)\n    \n    # Now 'predictions' contains the predictions for all images in the batch\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:53:45.702790Z","iopub.execute_input":"2024-05-23T08:53:45.703143Z","iopub.status.idle":"2024-05-23T08:53:45.710449Z","shell.execute_reply.started":"2024-05-23T08:53:45.703115Z","shell.execute_reply":"2024-05-23T08:53:45.709516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!cp '/kaggle/input/birdclef24-spectr-imagenettrained-model/birdclef2024_imagenet.keras' .\n\nmodel = tf.keras.models.load_model('birdclef2024_imagenet.keras')\n\npil_image = Image.open(\"/kaggle/input/birdclef-2024-mel-spectrograms/train_images/asbfly/XC164848_00.png\")\nprint(pil_image)\n# Make a prediction\npredictions = make_prediction(pil_image)\n\nprint(\"Predictions:\", predictions)","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:55:51.517073Z","iopub.execute_input":"2024-05-23T08:55:51.517543Z","iopub.status.idle":"2024-05-23T08:56:01.172359Z","shell.execute_reply.started":"2024-05-23T08:55:51.517513Z","shell.execute_reply":"2024-05-23T08:56:01.171131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#initialize with columns from sample_submission\nsample_submit = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\")\nsubmit = pd.DataFrame(columns=sample_submit.columns)\n\nsubmit","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:56:12.646938Z","iopub.execute_input":"2024-05-23T08:56:12.647320Z","iopub.status.idle":"2024-05-23T08:56:12.681748Z","shell.execute_reply.started":"2024-05-23T08:56:12.647271Z","shell.execute_reply":"2024-05-23T08:56:12.680827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_count_for_test_mode = 5\n\n#determine filenames\nfilenames_with_path = glob.glob(f\"{soundscapes_folder}/*.ogg\")\nfilenames = [os.path.basename(filename) for filename in filenames_with_path]\n\nfiles_handled = 0\nstart_time = time.time()\n\nfor filename in filenames:\n    #generate array of spectrograms for each file\n    images = get_spectrograms_for_ogg(soundscapes_folder, filename)\n    \n    time_index = 0\n\n    #predict for all images\n    prediction_batch_results = make_prediction_batch(images)\n    \n    #predictions to DF\n    for predictions in prediction_batch_results:\n        print(\".\", end=\"\")\n        filename_no_prefix = filename.replace(\".ogg\", \"\")\n\n        # Flatten predictions if necessary\n        predictions = predictions.flatten()\n\n        #make same prediction for multiple \n        for duplicate_pred_index in range(0, duplicate_predictions_count):\n            # Create a new row dictionary with 'row_id' and prediction values\n            time_index += audio_duration\n            row_id = f\"{filename_no_prefix}_{int(time_index)}\"\n            new_row_dict = {'row_id': row_id}\n            for i, col_name in enumerate(submit.columns[1:]):  # Skip 'row_id' column\n                new_row_dict[col_name] = predictions[i]\n\n            # Convert the new row dictionary to a DataFrame\n            new_row_df = pd.DataFrame(new_row_dict, index=[0])\n\n            submit = pd.concat([submit, new_row_df], ignore_index=True)\n\n    #exit after first file processed if just doing a quick test    \n    files_handled += 1\n    if quick_test and file_count_for_test_mode == files_handled: break\n\n#Time estimate (need to be under 2 hours!)\nprint(f\"\\n{files_handled} files processed in {(time.time() - start_time)} seconds\")\navg_time = ((time.time() - start_time) / files_handled)\nprint(f\"Time / file: {avg_time} seconds\")\nexpected_files = 1100\ntotal_hours = (avg_time * expected_files / 3600)\nprint(f\"Estimated time for {expected_files}: {total_hours} hours\")","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:56:27.933983Z","iopub.execute_input":"2024-05-23T08:56:27.934681Z","iopub.status.idle":"2024-05-23T08:56:38.071435Z","shell.execute_reply.started":"2024-05-23T08:56:27.934650Z","shell.execute_reply":"2024-05-23T08:56:38.070366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv('submission.csv', index=False)\nsubmit","metadata":{"execution":{"iopub.status.busy":"2024-05-23T08:56:56.064266Z","iopub.execute_input":"2024-05-23T08:56:56.064969Z","iopub.status.idle":"2024-05-23T08:56:56.167458Z","shell.execute_reply.started":"2024-05-23T08:56:56.064938Z","shell.execute_reply":"2024-05-23T08:56:56.166508Z"},"trusted":true},"execution_count":null,"outputs":[]}]}