{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":8033468,"sourceType":"datasetVersion","datasetId":4735360}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\n\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nfrom tensorflow.keras import layers, models\n\nfrom tensorflow.keras.utils import to_categorical\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import GlobalAveragePooling2D, Dense\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.preprocessing import image\n\nfrom PIL import Image\nfrom IPython.display import display\n\nAUTOTUNE = tf.data.experimental.AUTOTUNE","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-05-19T23:27:45.626894Z","iopub.execute_input":"2024-05-19T23:27:45.627719Z","iopub.status.idle":"2024-05-19T23:27:45.634640Z","shell.execute_reply.started":"2024-05-19T23:27:45.627687Z","shell.execute_reply":"2024-05-19T23:27:45.633493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# A simple attempt at fine-tuning an ImageNet on Mel Spectrograms generated by:\n#### https://www.kaggle.com/code/richolson/birdclef2024-simple-mel-spectrogram-generator/notebook?scriptVersionId=170393744\n\n# For example of inference / submitting this model see:\n#### https://www.kaggle.com/code/richolson/birdclef-2024-spectrograms-imagenet-run","metadata":{}},{"cell_type":"markdown","source":"# Group spectrograms by the sound sample they came from\n* Each OGG file was potentially split into multiple spectrograms (of fixed time length)\n* We need to make sure samples from the same OGG file aren't split between train / test (or model could appear to perform much better than it does)","metadata":{}},{"cell_type":"code","source":"# Base directory where the folders are stored\nimage_folder = \"/kaggle/input/birdclef-2024-mel-spectrograms/train_images/\"\n\n# Mapping of classes to numerical labels\n# Classes are alphabetically sorted - so we can easily restore class order when we load\nclass_labels = {class_name: i for i, class_name in enumerate(sorted(os.listdir(image_folder)))}\nnum_classes = len(class_labels)\n\n# Collect all file paths and their corresponding class labels\nfile_paths = []\nlabels = []\n\n# New structure to keep track of groups\nsamples = {}\n\nfor class_name in os.listdir(image_folder):\n    class_dir = os.path.join(image_folder, class_name)\n    for filename in os.listdir(class_dir):\n        # Extract base sample name from filename (files split at \"_\")\n        sample_base = filename.split('_')[0] \n        full_path = os.path.join(class_dir, filename)\n        \n        if sample_base not in samples:\n            samples[sample_base] = {'files': [], 'label': class_labels[class_name]}\n        samples[sample_base]['files'].append(full_path)","metadata":{"execution":{"iopub.status.busy":"2024-05-19T23:27:45.636883Z","iopub.execute_input":"2024-05-19T23:27:45.637414Z","iopub.status.idle":"2024-05-19T23:27:45.820385Z","shell.execute_reply.started":"2024-05-19T23:27:45.637387Z","shell.execute_reply":"2024-05-19T23:27:45.819465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image pre-processing","metadata":{}},{"cell_type":"code","source":"def preprocess_image(file_path, label):\n    img = tf.io.read_file(file_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.random_contrast(img, lower=0.2, upper=1.8)\n    img = tf.image.random_brightness(img, max_delta=0.2)\n\n    #scale 0-1\n    img = tf.cast(img, tf.float32) / 255.0\n    \n    label = tf.one_hot(label, depth=num_classes)\n    return img, label","metadata":{"execution":{"iopub.status.busy":"2024-05-19T23:41:52.708104Z","iopub.execute_input":"2024-05-19T23:41:52.708542Z","iopub.status.idle":"2024-05-19T23:41:52.715372Z","shell.execute_reply.started":"2024-05-19T23:41:52.708501Z","shell.execute_reply":"2024-05-19T23:41:52.714376Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split our samples into train / test","metadata":{}},{"cell_type":"code","source":"test_size = 0.2\nbatch_size = 32\n\n# Convert the samples dictionary into a list for splitting\nsamples_list = list(samples.items())\n\n# Split the list of tuples into training and validation sets\ntrain_samples, val_samples = train_test_split(samples_list, test_size=test_size, random_state=42)\n\n# Function to extract file paths and labels from the split samples\ndef extract_files_and_labels(sample_list):\n    file_paths = []\n    labels = []\n    for _, sample_info in sample_list:\n        file_paths.extend(sample_info['files'])\n        labels.extend([sample_info['label']] * len(sample_info['files']))\n    return file_paths, labels\n\ntrain_files, train_labels = extract_files_and_labels(train_samples)\nval_files, val_labels = extract_files_and_labels(val_samples)\n\nnum_classes = np.max(train_labels) + 1\n\n# Create a tf.data.Dataset from file paths and labels\ntrain_dataset = tf.data.Dataset.from_tensor_slices((train_files, train_labels))\ntrain_dataset = train_dataset.map(preprocess_image, num_parallel_calls=AUTOTUNE)\ntrain_dataset = train_dataset.shuffle(buffer_size=1000).batch(batch_size).prefetch(AUTOTUNE)\n\nval_dataset = tf.data.Dataset.from_tensor_slices((val_files, val_labels))\nval_dataset = val_dataset.map(preprocess_image, num_parallel_calls=AUTOTUNE)\nval_dataset = val_dataset.batch(batch_size).prefetch(AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2024-05-19T23:27:45.832973Z","iopub.execute_input":"2024-05-19T23:27:45.833602Z","iopub.status.idle":"2024-05-19T23:27:46.118040Z","shell.execute_reply.started":"2024-05-19T23:27:45.833550Z","shell.execute_reply":"2024-05-19T23:27:46.117018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Define model","metadata":{}},{"cell_type":"code","source":"#MobileNetV2\nfrom tensorflow.keras.applications import MobileNetV2\nbase_model = MobileNetV2(weights='imagenet', include_top=False, input_shape=(224, 224, 3))\n\n# allow training of base model (worked better)\nbase_model.trainable = True\n\nx = base_model.output\nx = GlobalAveragePooling2D()(x)\n\n# Add a fully-connected layer\nx = Dense(1024, activation='relu')(x)\n\n# And a logistic layer matching class count\npredictions = Dense(num_classes, activation='softmax')(x)\n\n# Learning rate scheduler\ninitial_learning_rate = 0.00005\nlr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n    initial_learning_rate=initial_learning_rate,\n    decay_steps=1000,\n    decay_rate=0.96,\n    staircase=True)\n\noptimizer = tf.keras.optimizers.Adam(learning_rate=lr_schedule)\n\n# This is the model we will train\nmodel = Model(inputs=base_model.input, outputs=predictions)\n\nmodel.compile(optimizer=optimizer,\n              loss='categorical_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2024-05-19T23:27:46.119283Z","iopub.execute_input":"2024-05-19T23:27:46.119573Z","iopub.status.idle":"2024-05-19T23:27:47.521876Z","shell.execute_reply.started":"2024-05-19T23:27:46.119548Z","shell.execute_reply":"2024-05-19T23:27:47.520813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train!","metadata":{}},{"cell_type":"code","source":"history = model.fit(\n    train_dataset,\n    epochs=15,\n    validation_data=val_dataset\n)","metadata":{"execution":{"iopub.status.busy":"2024-05-19T23:27:47.523128Z","iopub.execute_input":"2024-05-19T23:27:47.523507Z","iopub.status.idle":"2024-05-19T23:38:07.404200Z","shell.execute_reply.started":"2024-05-19T23:27:47.523468Z","shell.execute_reply":"2024-05-19T23:38:07.403070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Sanity check (make sure we at least predict something)\n* Verify we aren't having problem with model predicting the same value for all classes\n* Sure would be nice if it correctly identified species... (Asbfly)","metadata":{}},{"cell_type":"code","source":"def preprocess_image_pred(pil_img):\n    img = pil_img.convert('RGB')\n    img_array = image.img_to_array(img)\n    img_array = np.expand_dims(img_array, axis=0)\n    img_array /= 255.0\n    return img_array\n\ndef make_prediction(image):\n    img_array = preprocess_image_pred(image)\n    predictions = model.predict(img_array)\n        \n    return predictions\n\npil_image = Image.open(\"/kaggle/input/birdclef-2024-mel-spectrograms/train_images/asbfly/XC164848_00.png\")\ndisplay(pil_image)\n\npredictions = make_prediction(pil_image)\nprint(\"Asbfly: \", predictions[0][0])\nprint(\"\\nAll preds:\\n\", predictions[0])\n","metadata":{"execution":{"iopub.status.busy":"2024-05-19T23:48:12.227251Z","iopub.execute_input":"2024-05-19T23:48:12.227646Z","iopub.status.idle":"2024-05-19T23:48:12.327032Z","shell.execute_reply.started":"2024-05-19T23:48:12.227615Z","shell.execute_reply":"2024-05-19T23:48:12.326096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Save Model","metadata":{}},{"cell_type":"code","source":"model.save('birdclef2024_imagenet.keras')\n","metadata":{"execution":{"iopub.status.busy":"2024-05-19T23:38:11.961283Z","iopub.execute_input":"2024-05-19T23:38:11.961658Z","iopub.status.idle":"2024-05-19T23:38:12.706025Z","shell.execute_reply.started":"2024-05-19T23:38:11.961621Z","shell.execute_reply":"2024-05-19T23:38:12.705276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test Loading Model","metadata":{}},{"cell_type":"code","source":"new_model = tf.keras.models.load_model('birdclef2024_imagenet.keras')","metadata":{"execution":{"iopub.status.busy":"2024-05-19T23:38:12.707181Z","iopub.execute_input":"2024-05-19T23:38:12.707564Z","iopub.status.idle":"2024-05-19T23:38:15.311593Z","shell.execute_reply.started":"2024-05-19T23:38:12.707530Z","shell.execute_reply":"2024-05-19T23:38:15.310752Z"},"trusted":true},"execution_count":null,"outputs":[]}]}