{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8209908,"sourceType":"datasetVersion","datasetId":4865209},{"sourceId":172210423,"sourceType":"kernelVersion"}],"dockerImageVersionId":30683,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nimport os\nimport random\nimport time\nimport glob\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport PIL\nfrom PIL import Image\nPIL.Image.MAX_IMAGE_PIXELS = 933120000\n\nimport torch\nfrom torch import nn\nfrom torchvision import transforms\nfrom torch.utils.data import Dataset\nfrom torch.utils.data import DataLoader\nfrom torch.utils.data import random_split\nfrom torchinfo import summary\nimport torch.nn.functional as F  # Import functional module for softmax\nimport timm\n\nfrom torchvision.transforms import ToPILImage, Resize, ToTensor\n\nimport librosa\nfrom scipy.signal import butter, filtfilt","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2024-05-06T00:25:50.238055Z","iopub.execute_input":"2024-05-06T00:25:50.238807Z","iopub.status.idle":"2024-05-06T00:25:56.175780Z","shell.execute_reply.started":"2024-05-06T00:25:50.238767Z","shell.execute_reply":"2024-05-06T00:25:56.174246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Version 6 of this notebook scored .61 on the LB\n# Version 7 is unchanged except for a scoring experiment:\n* For each soundscape a number between 0 and 1 is added to all predictions for that soundscape.\n* If all soundscapes are scored together - the expectation is that this test will result in a significant reduction in score (LB close to .50)\n* If all soundscapes are scored separately (and then averaged) - this should result in no significant change to the score (LB close to .61)\n* <b>Result:</b> this version scored .50 - which is consistent with soundscapes being scored together\n\n# Version 8 is a slight variation of the scoring experiment:\n* For each soundscape all predictions are multiplied by a random number between 0.5 and 1.0\n* <b>Result:</b> this version scored .61! (that's curious...)\n\n# Version 9 is another variation:\n* For each soundscape all predictions are multiplied by a random number between 0.1 and 1.0\n\n## More information on scoring at:\nhttps://www.kaggle.com/code/richolson/birdclef-2024-exploring-scoring","metadata":{}},{"cell_type":"markdown","source":"# This notebook runs the ImageNet model we trained at:\nhttps://www.kaggle.com/code/richolson/birdclef-2024-train-v2/\n\n# Based on contiguous spectrograms we made here:\nhttps://www.kaggle.com/code/richolson/birdclef-2024-contiguous-mel-spectrogram-generator\n\n# The model was trained on 10 second segments of spectrogram scaled to 224x224\n* Seemed like bird calls frequently didn't appear in 5-second segments\n* Test data is convered into 10-second spectrogram images for prediction\n* Each prediction duplicated to the corresponding 2x 5-second periods for submission","metadata":{}},{"cell_type":"code","source":"#CPU for prediction...\ndevice = 'cpu'\n\n#bandpass filter for audio (Hz)\n#same parameters as training data\nlow_cut = 400\nhigh_cut = 10000\n\nimagenet_input_size = 224\nspectrogram_height = 224\nspectrogram_width_per_5sec = 512  #spectrograms initially generated at same resolution as originals (resize to 224 for 10-seconds)\n\n#we are training the model on sample segment this long\nmodel_sample_time_sec = 10\n\n#frequency that submissions are expected to predict at\nexpected_prediction_frequency = 5\n\n#predictions are duplicated this many rows\nduplicate_predictions_count = 2","metadata":{"execution":{"iopub.status.busy":"2024-05-06T00:25:56.177985Z","iopub.execute_input":"2024-05-06T00:25:56.178385Z","iopub.status.idle":"2024-05-06T00:25:56.186612Z","shell.execute_reply.started":"2024-05-06T00:25:56.178335Z","shell.execute_reply":"2024-05-06T00:25:56.184980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Paths / etc.","metadata":{}},{"cell_type":"code","source":"mel_dir = \"/kaggle/input/birdclef-2024-contiguous-mel-spectrogram-generator/train_images/\"\naudio_dir = \"/kaggle/input/birdclef-2024/train_audio/\"\n\nsample_submit = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\")\n\nsoundscapes_folder = \"/kaggle/input/birdclef-2024/test_soundscapes\"\nquick_test = False\n\n#if we don't have any files in test_soundscapes - revert to test mode\n\nif len(glob.glob(f\"{soundscapes_folder}/*.ogg\")) == 0:\n    soundscapes_folder = \"/kaggle/input/birdclef-2024/unlabeled_soundscapes\"\n    quick_test = True","metadata":{"execution":{"iopub.status.busy":"2024-05-06T00:25:56.188935Z","iopub.execute_input":"2024-05-06T00:25:56.189979Z","iopub.status.idle":"2024-05-06T00:25:56.216511Z","shell.execute_reply.started":"2024-05-06T00:25:56.189939Z","shell.execute_reply":"2024-05-06T00:25:56.215262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load model","metadata":{}},{"cell_type":"code","source":"# Define the path to the saved model\nmodel_path = \"/kaggle/input/birdclef-2024-train-v2/model.pth\"\n\n# Load the entire model\nmodel = torch.load(model_path, map_location=torch.device('cpu'))\n\n# Move the model to the desired device\nmodel.to(device)\nsummary(model)","metadata":{"execution":{"iopub.status.busy":"2024-05-06T00:25:56.218012Z","iopub.execute_input":"2024-05-06T00:25:56.218674Z","iopub.status.idle":"2024-05-06T00:25:56.387890Z","shell.execute_reply.started":"2024-05-06T00:25:56.218633Z","shell.execute_reply":"2024-05-06T00:25:56.386753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image handling setup","metadata":{}},{"cell_type":"code","source":"resize_transform = transforms.Compose([\n    transforms.Resize((imagenet_input_size, imagenet_input_size)),  # Resize the image to 224x224\n    transforms.Grayscale(num_output_channels=3),  # Convert grayscale to RGB by replicating channels\n    transforms.ToTensor(),  # Convert the image to a PyTorch tensor\n])","metadata":{"execution":{"iopub.status.busy":"2024-05-06T00:25:56.391211Z","iopub.execute_input":"2024-05-06T00:25:56.391593Z","iopub.status.idle":"2024-05-06T00:25:56.399015Z","shell.execute_reply.started":"2024-05-06T00:25:56.391563Z","shell.execute_reply":"2024-05-06T00:25:56.397598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Function to handle inference\n* This generates spectrograms for test data / predicts on them","metadata":{}},{"cell_type":"code","source":"def bandpass_filter(data, lowcut, highcut, sr, order=5):\n    nyquist = 0.5 * sr\n    low = lowcut / nyquist\n    high = highcut / nyquist\n    b, a = butter(order, [low, high], btype='band')\n    y = filtfilt(b, a, data)\n    return y\n\n#using same hope length as dataset creation (rescale output to 224x224)\nhop_length = int((160400 / spectrogram_width_per_5sec))\n\ndef evaluate_audio_file_segments(audio_path, return_images = False):\n    \n    start_time = time.time()\n\n    # Load the audio file\n    audio, sr = librosa.load(audio_path, sr=None)\n    audio = bandpass_filter(audio, low_cut, high_cut, sr)\n\n    # Calculate the number of samples per segment\n    samples_per_segment = sr * model_sample_time_sec\n\n    # Split the audio into segments\n    total_samples = len(audio)\n    segments = [audio[i:i + samples_per_segment] for i in range(0, total_samples, samples_per_segment) if i + samples_per_segment <= total_samples]\n\n    predictions = []  # Store predictions for each segment\n    total_time = 0\n    \n    images = []\n    \n    for segment in segments:\n        # Process each segment into a spectrogram\n        spectrogram = librosa.feature.melspectrogram(y=segment, sr=sr, hop_length=hop_length, n_mels=imagenet_input_size, fmin=low_cut, fmax=high_cut)\n        spectrogram_db = librosa.amplitude_to_db(spectrogram, ref=np.max)\n\n        # Normalize spectrogram for image display\n        spectrogram_norm = (spectrogram_db - spectrogram_db.min()) / (spectrogram_db.max() - spectrogram_db.min()) * 255\n        spectrogram_image = Image.fromarray(spectrogram_norm.astype(np.uint8))                \n                \n        # Convert the PIL Image to a tensor\n        spectrogram_image_tensor = resize_transform(spectrogram_image)\n        spectrogram_image_tensor = spectrogram_image_tensor.to(device)\n\n        model.eval()\n        \n        with torch.no_grad():\n            final_tensor = spectrogram_image_tensor.repeat(1, 1, 1, 1)\n            logits = model(final_tensor)  # Assuming model expects 3-channel input\n\n        probabilities = F.softmax(logits, dim=1)  # Apply softmax to convert logits to probabilities\n        predictions.append(probabilities.cpu().numpy())\n\n        # Convert the tensor back to a PIL Image (just for visually verifying image sent to model was good)\n        if return_images:\n            image_pil = transforms.ToPILImage()(spectrogram_image_tensor)\n            images.append(image_pil)\n\n\n    print (\"Time per file:\", time.time()- start_time)\n        \n    return predictions, images","metadata":{"execution":{"iopub.status.busy":"2024-05-06T00:25:56.401329Z","iopub.execute_input":"2024-05-06T00:25:56.401866Z","iopub.status.idle":"2024-05-06T00:25:56.420568Z","shell.execute_reply.started":"2024-05-06T00:25:56.401827Z","shell.execute_reply":"2024-05-06T00:25:56.418781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Verify predictions and images look good\n* This tends to come back really slow the first time - so just re-run...\n* We are testing on a sample file for the \"Asbfly\" - if all is working well - the first prediction column should have a larger than other columns\n* This is a small file - so this cell isn't a good indication of runtime","metadata":{}},{"cell_type":"code","source":"predictions, images = evaluate_audio_file_segments(\"/kaggle/input/birdclef-2024/train_audio/asbfly/XC267680.ogg\", return_images = True)\n\n#make sure predictions and images look OK\ndisplay(images[0])\nprint(\"Odds for Asbfly:\", predictions[0][0][0])","metadata":{"execution":{"iopub.status.busy":"2024-05-06T00:25:56.422608Z","iopub.execute_input":"2024-05-06T00:25:56.423009Z","iopub.status.idle":"2024-05-06T00:25:59.661691Z","shell.execute_reply.started":"2024-05-06T00:25:56.422976Z","shell.execute_reply":"2024-05-06T00:25:59.660331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Initialize DF","metadata":{}},{"cell_type":"code","source":"#initialize with columns from sample_submission\nsample_submit = pd.read_csv(\"/kaggle/input/birdclef-2024/sample_submission.csv\")\nsubmit = pd.DataFrame(columns=sample_submit.columns)\nsubmit","metadata":{"execution":{"iopub.status.busy":"2024-05-06T00:26:54.028147Z","iopub.execute_input":"2024-05-06T00:26:54.028641Z","iopub.status.idle":"2024-05-06T00:26:54.066418Z","shell.execute_reply.started":"2024-05-06T00:26:54.028605Z","shell.execute_reply":"2024-05-06T00:26:54.064908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Predict on test data / inference time test\n* Runs on first 5 files in unlabeled_soundscapes if test data not present\n* Prediction time per file should be < 6.5 seconds in order to complete in < 2 hours on test data","metadata":{}},{"cell_type":"code","source":"files_for_quick_test = 5\n\nfilenames_with_path = glob.glob(f\"{soundscapes_folder}/*.ogg\")\nfilenames = [os.path.basename(filename) for filename in filenames_with_path]\n\nfiles_evaluated = 0\nfor filename in filenames:\n    start_time = time.time()\n    \n    predictions, images = evaluate_audio_file_segments(f\"{soundscapes_folder}/{filename}\", return_images = False)\n    \n    time_index = 0\n    \n    #decide on a random multiplier for each entry this soundscape\n    random_multiplier_per_soundscape = .1 + (random.random() * .9)\n    print(\"Randomizer for this soundscape:\", random_multiplier_per_soundscape)\n\n    #predictions to DF\n    for predictions in predictions:\n        filename_no_prefix = filename.replace(\".ogg\", \"\")\n\n        predictions = predictions.flatten()\n\n        #make same prediction for multiple \n        for duplicate_pred_index in range(0, duplicate_predictions_count):\n            # Create a new row dictionary with 'row_id' and prediction values\n            time_index += expected_prediction_frequency\n            row_id = f\"{filename_no_prefix}_{int(time_index)}\"\n            new_row_dict = {'row_id': row_id}\n            \n            for i, col_name in enumerate(submit.columns[1:]):  # Skip 'row_id' column\n\n                #add random number to each pred this soundscape\n                new_row_dict[col_name] = predictions[i] * random_multiplier_per_soundscape\n\n            # Convert the new row dictionary to a DataFrame\n            new_row_df = pd.DataFrame(new_row_dict, index=[0])\n\n            submit = pd.concat([submit, new_row_df], ignore_index=True)\n            \n    files_evaluated += 1\n            \n    if quick_test and files_evaluated == files_for_quick_test: break","metadata":{"execution":{"iopub.status.busy":"2024-05-06T00:30:31.095512Z","iopub.execute_input":"2024-05-06T00:30:31.095959Z","iopub.status.idle":"2024-05-06T00:31:00.961136Z","shell.execute_reply.started":"2024-05-06T00:30:31.095929Z","shell.execute_reply":"2024-05-06T00:31:00.959670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submit!","metadata":{}},{"cell_type":"code","source":"submit.to_csv('submission.csv', index=False)\nsubmit","metadata":{"execution":{"iopub.status.busy":"2024-05-06T00:27:30.993014Z","iopub.execute_input":"2024-05-06T00:27:30.993491Z","iopub.status.idle":"2024-05-06T00:27:31.138960Z","shell.execute_reply.started":"2024-05-06T00:27:30.993457Z","shell.execute_reply":"2024-05-06T00:27:31.137602Z"},"trusted":true},"execution_count":null,"outputs":[]}]}