{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8033468,"sourceType":"datasetVersion","datasetId":4735360},{"sourceId":8462031,"sourceType":"datasetVersion","datasetId":4773478},{"sourceId":8688117,"sourceType":"datasetVersion","datasetId":5209242}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Running the ImageNet model we trained at:\n### https://www.kaggle.com/code/richolson/birdclef-2024-spectrograms-imagenet-train\n* Published via this dataset: https://www.kaggle.com/datasets/richolson/birdclef24-spectr-imagenettrained-model\n* Note: This version of notebook was tested using version 4 of the above dataset (other versions may not perform well)\n\n## ...with the spectrograms we generated here:\n### https://www.kaggle.com/code/richolson/birdclef2024-simple-mel-spectrogram-generator\n\n## This version is using MobileNetV2 with backbone training\n* Backbone training seems to have had a significant negative impact on model performance - so....\n* Using approach of predicting every 15-seconds (and duplicating answers)\n","metadata":{}},{"cell_type":"code","source":"import os\nimport io\nimport glob\nimport pandas as pd\nimport numpy as np\nimport tensorflow as tf\nimport time\n\nfrom PIL import Image, ImageOps\n\nfrom tensorflow.keras.preprocessing import image\n\nimport librosa\nimport librosa.display\n\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import load_model\n\nimport matplotlib.cm as cm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-13T10:45:29.444333Z","iopub.execute_input":"2024-06-13T10:45:29.445224Z","iopub.status.idle":"2024-06-13T10:45:45.917902Z","shell.execute_reply.started":"2024-06-13T10:45:29.445185Z","shell.execute_reply":"2024-06-13T10:45:45.916645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Settings","metadata":{}},{"cell_type":"markdown","source":"settings and configuration data ","metadata":{}},{"cell_type":"code","source":"train_meta = pd.read_csv(\"/kaggle/input/birdclef-2024/train_metadata.csv\")\n\nsoundscapes_folder = \"/kaggle/input/birdclef-2024/test_soundscapes\"\nquick_test = False\n\n#if we don't have any files in test_soundscapes - revert to test mode\n\nif len(glob.glob(f\"{soundscapes_folder}/*.ogg\")) == 0:\n    soundscapes_folder = \"/kaggle/input/birdclef-2024/unlabeled_soundscapes\"\n    quick_test = True\n\n#spectrogram length\naudio_duration = 5\n\n#dimension of spectrograms\nimage_size = 224\n\n#we make the same predictions for this number of time indexes (for performance)\n#ie - value of 3 means making one prediction per 15 seconds - and then duplicating for the rest\n#1 = no skipping\nduplicate_predictions_count = 3","metadata":{"execution":{"iopub.status.busy":"2024-06-13T10:45:45.919979Z","iopub.execute_input":"2024-06-13T10:45:45.920661Z","iopub.status.idle":"2024-06-13T10:45:46.133860Z","shell.execute_reply.started":"2024-06-13T10:45:45.920624Z","shell.execute_reply":"2024-06-13T10:45:46.132640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Generate Spectrograms for OGG file\n## Generates array of images of spectrograms from test data so we can use ImageNet to predict on them","metadata":{}},{"cell_type":"markdown","source":"test data(audio file is given as input which will generate spectrogram images ","metadata":{}},{"cell_type":"code","source":"def get_spectrograms_for_ogg(dirname, filename):\n\n    images = []\n    \n    # Load the entire audio file once\n    audio_path = os.path.join(dirname, filename)\n    audio_data, sr = librosa.load(audio_path, sr=None)\n    total_duration = librosa.get_duration(y=audio_data, sr=sr)\n    \n    num_segments = int(total_duration // audio_duration)\n\n    for segment in range(0, num_segments, duplicate_predictions_count):\n        offset_samples = int(segment * audio_duration * sr)\n        end_samples = int(offset_samples + audio_duration * sr)\n        \n        segment_data = audio_data[offset_samples:end_samples]\n        \n        # Generate the spectrogram\n\n        S = librosa.feature.melspectrogram(y=segment_data, sr=sr, n_mels=128)\n        S_db = librosa.amplitude_to_db(S, ref=np.max)\n                \n        #convert spectrogram data into directly into image (much faster than matplotlib)\n        normalized_array = (S_db - np.min(S_db)) / (np.max(S_db) - np.min(S_db))\n        \n        #set color mapping (so consistent with train)\n        spectrogram_image = cm.magma(normalized_array)[:, :, :3]\n        spectrogram_image = (spectrogram_image * 255).astype(np.uint8)\n        \n        spectrogram_image = Image.fromarray(spectrogram_image)\n        \n        #resize and flip (so consistent with train)\n        spectrogram_image = spectrogram_image.resize((image_size, image_size), Image.ANTIALIAS)\n        spectrogram_image = ImageOps.flip(spectrogram_image)\n\n        \n        images.append(spectrogram_image)\n\n    return images\n","metadata":{"execution":{"iopub.status.busy":"2024-06-13T10:45:46.139785Z","iopub.execute_input":"2024-06-13T10:45:46.140133Z","iopub.status.idle":"2024-06-13T10:45:46.153114Z","shell.execute_reply.started":"2024-06-13T10:45:46.140104Z","shell.execute_reply":"2024-06-13T10:45:46.151934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Verify spectrograms are being generated","metadata":{}},{"cell_type":"markdown","source":"demo to verify the spectrograms generayed has thesignal and not noise ","metadata":{}},{"cell_type":"code","source":"images = get_spectrograms_for_ogg(\"/kaggle/input/birdclef-2024/unlabeled_soundscapes/\", \"1000170626.ogg\")\nfor image_index in range(0,3):\n    display(images[image_index])","metadata":{"execution":{"iopub.status.busy":"2024-06-13T10:45:46.154460Z","iopub.execute_input":"2024-06-13T10:45:46.154876Z","iopub.status.idle":"2024-06-13T10:46:07.675143Z","shell.execute_reply.started":"2024-06-13T10:45:46.154844Z","shell.execute_reply":"2024-06-13T10:46:07.674065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data loaders","metadata":{}},{"cell_type":"raw","source":"","metadata":{}},{"cell_type":"raw","source":"","metadata":{}},{"cell_type":"markdown","source":"predictions are done as per batches","metadata":{}},{"cell_type":"code","source":"def preprocess_image(pil_img):\n    # Convert the image to RGB mode in case it's not\n    img = pil_img.convert('RGB')\n        \n    # Convert the PIL image to a numpy array\n    img_array = image.img_to_array(img)\n    \n    # Expand dimensions to have shape (1, 224, 224, 3)\n    img_array = np.expand_dims(img_array, axis=0)\n    \n    # Scale pixel values to [0, 1]\n    img_array /= 255.0\n    \n    return img_array\n\ndef make_prediction(image):\n    #classes were created in alphabetical order so can be loaded into DF with alpha column order\n    img_array = preprocess_image(image)\n    predictions = model.predict(img_array, verbose=None)    \n    return predictions\n\n\ndef make_prediction_batch(images):\n    # Preprocess each image and stack them into a single batch tensor\n    img_batch = np.stack([preprocess_image(image) for image in images])\n        \n    # Predict on the batch\n    predictions = model.predict(np.squeeze(img_batch), verbose=None)\n    \n    # Now 'predictions' contains the predictions for all images in the batch\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2024-06-13T10:46:07.677103Z","iopub.execute_input":"2024-06-13T10:46:07.678280Z","iopub.status.idle":"2024-06-13T10:46:07.688767Z","shell.execute_reply.started":"2024-06-13T10:46:07.678241Z","shell.execute_reply":"2024-06-13T10:46:07.687641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load our model / verify we can predict\n### Earlier had problems with model loading and predicting same value of all classes","metadata":{}},{"cell_type":"markdown","source":"speed was slow and was taking longer time ","metadata":{}},{"cell_type":"code","source":"#keras needs to unzip from read-write file space\n!cp '/kaggle/input/birdclef24-spectr-imagenettrained-model/birdclef2024_imagenet.keras' .\n\nmodel = tf.keras.models.load_model('birdclef2024_imagenet.keras')\n\npil_image = Image.open(\"/kaggle/input/birdclef-2024-mel-spectrograms/train_images/asbfly/XC164848_00.png\")\nprint(pil_image)\n# Make a prediction\npredictions = make_prediction(pil_image)\n\nprint(\"Predictions:\", predictions)","metadata":{"execution":{"iopub.status.busy":"2024-06-13T10:46:07.690945Z","iopub.execute_input":"2024-06-13T10:46:07.693179Z","iopub.status.idle":"2024-06-13T10:46:16.785303Z","shell.execute_reply.started":"2024-06-13T10:46:07.693127Z","shell.execute_reply":"2024-06-13T10:46:16.783663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Initialize submit DF with correct columns","metadata":{}},{"cell_type":"code","source":"#initialize with columns from sample_submission\nsample_submit = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\")\nsubmit = pd.DataFrame(columns=sample_submit.columns)\n\nsubmit","metadata":{"execution":{"iopub.status.busy":"2024-06-13T10:46:16.787467Z","iopub.execute_input":"2024-06-13T10:46:16.787959Z","iopub.status.idle":"2024-06-13T10:46:16.844220Z","shell.execute_reply.started":"2024-06-13T10:46:16.787915Z","shell.execute_reply":"2024-06-13T10:46:16.843092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict for soundscapes","metadata":{}},{"cell_type":"code","source":"#determine filenames\nfilenames_with_path = glob.glob(f\"{soundscapes_folder}/*.ogg\")\nfilenames = [os.path.basename(filename) for filename in filenames_with_path]\n\nfor filename in filenames:\n    start_time = time.time()\n    #generate array of spectrograms for each file\n    images = get_spectrograms_for_ogg(soundscapes_folder, filename)\n    \n    time_index = 0\n\n    #predict for all images\n    prediction_batch_results = make_prediction_batch(images)\n    \n    #predictions to DF\n    for predictions in prediction_batch_results:\n        print(\".\", end=\"\")\n        filename_no_prefix = filename.replace(\".ogg\", \"\")\n\n        # Flatten predictions if necessary\n        predictions = predictions.flatten()\n\n        #make same prediction for multiple \n        for duplicate_pred_index in range(0, duplicate_predictions_count):\n            # Create a new row dictionary with 'row_id' and prediction values\n            time_index += audio_duration\n            row_id = f\"{filename_no_prefix}_{int(time_index)}\"\n            new_row_dict = {'row_id': row_id}\n            for i, col_name in enumerate(submit.columns[1:]):  # Skip 'row_id' column\n                new_row_dict[col_name] = predictions[i]\n\n            # Convert the new row dictionary to a DataFrame\n            new_row_df = pd.DataFrame(new_row_dict, index=[0])\n\n            submit = pd.concat([submit, new_row_df], ignore_index=True)\n        \n    #needs to be <6.5 seconds to handle 1100 files in 7200 seconds (2 hours)\n    #(this may run a bit slow the first time)\n    print(f\"\\nTime to process file {time.time() - start_time}\")\n    #exit after first file processed if just doing a quick test\n    if quick_test: break\n        ","metadata":{"execution":{"iopub.status.busy":"2024-06-13T10:46:16.845949Z","iopub.execute_input":"2024-06-13T10:46:16.846624Z","iopub.status.idle":"2024-06-13T10:46:27.359243Z","shell.execute_reply.started":"2024-06-13T10:46:16.846587Z","shell.execute_reply":"2024-06-13T10:46:27.358054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Find species that aren't in Western Ghats in training data","metadata":{}},{"cell_type":"code","source":"\ndef get_non_western_ghat_species():\n    #Western Ghats\n    lower_latitude = 8\n    upper_latitude = 22\n    lower_longitude = 73\n    upper_longitude = 79\n\n    wg_train = train_meta.copy()\n\n    #filter train meta data to western ghats\n    wg_train = wg_train[(train_meta['latitude'] >= lower_latitude) & \n                               (train_meta['latitude'] <= upper_latitude) &\n                               (train_meta['longitude'] >= lower_longitude) &\n                               (train_meta['longitude'] <= upper_longitude)]\n\n    all_species = set(train_meta['primary_label'])\n    wg_species = set(wg_train['primary_label'])\n\n    # list of species only found in samples outside of wg\n    species_not_in_wg_primary = list(all_species - wg_species)\n\n    # check if any of those species are found as secondary labels for an wg samples\n    species_in_wg_secondary = [name for name in species_not_in_wg_primary if any(name in sublist for sublist in wg_train['secondary_labels'])]\n\n    #remove any species found as secondary labels\n    non_wg_species = set(species_not_in_wg_primary) - set(species_in_wg_secondary)\n\n    return non_wg_species","metadata":{"execution":{"iopub.status.busy":"2024-06-13T10:46:27.363470Z","iopub.execute_input":"2024-06-13T10:46:27.364310Z","iopub.status.idle":"2024-06-13T10:46:27.374269Z","shell.execute_reply.started":"2024-06-13T10:46:27.364266Z","shell.execute_reply":"2024-06-13T10:46:27.373103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Set non-Western Ghat species predictions to 0","metadata":{}},{"cell_type":"code","source":"non_wg_species = get_non_western_ghat_species()\nfor column in submit.columns:\n    if column in non_wg_species:\n        submit[column] = 0","metadata":{"execution":{"iopub.status.busy":"2024-06-13T10:46:27.375635Z","iopub.execute_input":"2024-06-13T10:46:27.376017Z","iopub.status.idle":"2024-06-13T10:46:27.423886Z","shell.execute_reply.started":"2024-06-13T10:46:27.375988Z","shell.execute_reply":"2024-06-13T10:46:27.422610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submit!","metadata":{}},{"cell_type":"markdown","source":"only the relevant files in westrern ghats region is used for prediction","metadata":{}},{"cell_type":"code","source":"submit.to_csv('submission.csv', index=False)\nsubmit","metadata":{"execution":{"iopub.status.busy":"2024-06-13T10:46:27.425275Z","iopub.execute_input":"2024-06-13T10:46:27.425649Z","iopub.status.idle":"2024-06-13T10:46:27.510520Z","shell.execute_reply.started":"2024-06-13T10:46:27.425615Z","shell.execute_reply":"2024-06-13T10:46:27.509322Z"},"trusted":true},"execution_count":null,"outputs":[]}]}