{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":70203,"databundleVersionId":8068726}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport glob\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers, models, callbacks\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\n\n\n#  CONFIGURATION & HYPERPARAMETERS\n\nBATCH_SIZE = 32\nIMG_HEIGHT = 224   \nIMG_WIDTH  = 224   \nDURATION   = 5     \nSR         = 22050\nEPOCHS     = 15\n\n#LOAD AND PREPARE METADATA\n\nmeta_file = glob.glob('/kaggle/input/**/train_metadata.csv', recursive=True)[0]\nDATASET_PATH = os.path.dirname(meta_file)\naudio_folder = f\"{DATASET_PATH}/train_audio\"\n\nmetadata = pd.read_csv(meta_file)\n\n# Encode Labels (Strings -> Integers)\nlabel_encoder = LabelEncoder()\nmetadata['target'] = label_encoder.fit_transform(metadata['primary_label'])\nNUM_CLASSES = len(label_encoder.classes_)\n\n# Split Metadata into Train and Validation\n\ntrain_df, val_df = train_test_split(\n    metadata, \n    test_size=0.2, \n    random_state=42, \n    stratify=metadata['target']\n)\n\n\n#  THE DATA GENERATOR (Memory Safe Streaming)\n\nclass BirdDataGenerator(keras.utils.Sequence):\n    def __init__(self, df, audio_dir, batch_size=32, shuffle=True):\n        self.df = df.reset_index(drop=True)\n        self.audio_dir = audio_dir\n        self.batch_size = batch_size\n        self.shuffle = shuffle\n        self.indexes = np.arange(len(self.df))\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n\n    def __len__(self):\n        return int(np.ceil(len(self.df) / self.batch_size))\n\n    def on_epoch_end(self):\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n\n    def __getitem__(self, index):\n        batch_indexes = self.indexes[index * self.batch_size : (index + 1) * self.batch_size]\n        batch_df = self.df.iloc[batch_indexes]\n\n        X = np.empty((len(batch_df), IMG_HEIGHT, IMG_WIDTH, 1), dtype=np.float32)\n        y = np.empty((len(batch_df)), dtype=np.int32)\n\n        for i, (_, row) in enumerate(batch_df.iterrows()):\n            filepath = os.path.join(self.audio_dir, row['filename'])\n            \n            try:\n                audio, sr = librosa.load(filepath, sr=SR, duration=DURATION, mono=True)\n                target_length = SR * DURATION\n                \n                # Pad short audio or truncate long audio\n                if len(audio) < target_length:\n                    audio = np.pad(audio, (0, target_length - len(audio)))\n                else:\n                    audio = audio[:target_length]\n                    \n                # Convert to Mel Spectrogram\n                spec = librosa.feature.melspectrogram(y=audio, sr=sr, n_mels=IMG_HEIGHT)\n                spec_db = librosa.power_to_db(spec, ref=np.max)\n                spec_resized = librosa.util.fix_length(spec_db, size=IMG_WIDTH, axis=1)\n                \n                # Normalize between 0 and 1\n                spec_norm = (spec_resized - spec_resized.min()) / (spec_resized.max() - spec_resized.min() + 1e-6)\n                \n                X[i,] = np.expand_dims(spec_norm, axis=-1)\n                y[i] = row['target']\n            except Exception as e:\n                # Fallback for corrupted files to prevent crashes\n                X[i,] = np.zeros((IMG_HEIGHT, IMG_WIDTH, 1))\n                y[i] = row['target']\n\n        return X, y\n\n# Instantiate Generators\ntrain_gen = BirdDataGenerator(train_df, audio_folder, batch_size=BATCH_SIZE, shuffle=True)\nval_gen = BirdDataGenerator(val_df, audio_folder, batch_size=BATCH_SIZE, shuffle=False)\n\n#  BUILD THE EFFICIENTNET MODEL\n\ninputs = layers.Input(shape=(IMG_HEIGHT, IMG_WIDTH, 1))\n\n# 1. Convert 1 channel to 3 channels dynamically inside the network\nx = layers.Conv2D(3, 1, padding='same', use_bias=False, activation='relu')(inputs)\nx = layers.BatchNormalization()(x)\n\n# 2. Load EfficientNet Backbone INDEPENDENTLY\nbackbone = keras.applications.EfficientNetV2B0(\n    include_top=False,\n    weights='imagenet', \n    input_shape=(IMG_HEIGHT, IMG_WIDTH, 3)  # Tell it to expect 3 channels\n)\nbackbone.trainable = True\n\n# 3. Pass our custom 3-channel tensor THROUGH the backbone\nx = backbone(x)\n\n# 4. Custom Classification Head\nx = layers.GlobalAveragePooling2D(name='gap')(x)\nx = layers.BatchNormalization(name='head_bn1')(x)\nx = layers.Dropout(0.5, name='head_drop1')(x)\nx = layers.Dense(512, activation='swish', name='head_dense')(x)\nx = layers.BatchNormalization(name='head_bn2')(x)\nx = layers.Dropout(0.4, name='head_drop2')(x)\noutputs = layers.Dense(NUM_CLASSES, activation='softmax')(x)\n\nmodel = models.Model(inputs=inputs, outputs=outputs)\n\nmodel.compile(\n    optimizer=keras.optimizers.Adam(learning_rate=1e-3),\n    loss='sparse_categorical_crossentropy',\n    metrics=['accuracy']\n)\n\n\n\n\n\n#  TRAINING\n\ncallbacks_list = [\n    # Saving to Kaggle's working directory\n    callbacks.ModelCheckpoint('/kaggle/working/best_bird_model.keras', monitor='val_accuracy', save_best_only=True, verbose=1),\n    callbacks.EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True, verbose=1),\n    callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=2, min_lr=1e-6, verbose=1)\n]\n\nprint(\"🚀 Starting Training on Kaggle GPU...\")\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=EPOCHS,\n    callbacks=callbacks_list\n)\n\n# Save the training history graph data\npd.DataFrame(history.history).to_csv('/kaggle/working/training_history.csv')\nprint(\"✅ Training Complete!\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-05-18T16:01:43.386492Z","iopub.execute_input":"2026-05-18T16:01:43.387043Z"}},"outputs":[{"name":"stderr","text":"I0000 00:00:1779120115.292470      57 gpu_device.cc:2019] Created device /job:localhost/replica:0/task:0/device:GPU:0 with 13757 MB memory:  -> device: 0, name: Tesla T4, pci bus id: 0000:00:04.0, compute capability: 7.5\nI0000 00:00:1779120115.300338      57 gpu_device.cc:2019] Created device /job:localhost/replica:0/task:0/device:GPU:1 with 13757 MB memory:  -> device: 1, name: Tesla T4, pci bus id: 0000:00:05.0, compute capability: 7.5\n","output_type":"stream"},{"name":"stdout","text":"Downloading data from https://storage.googleapis.com/tensorflow/keras-applications/efficientnet_v2/efficientnetv2-b0_notop.h5\n\u001b[1m24274472/24274472\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m2s\u001b[0m 0us/step\n🚀 Starting Training on Kaggle GPU...\n","output_type":"stream"},{"name":"stderr","text":"/usr/local/lib/python3.12/dist-packages/keras/src/trainers/data_adapters/py_dataset_adapter.py:121: UserWarning: Your `PyDataset` class should call `super().__init__(**kwargs)` in its constructor. `**kwargs` can include `workers`, `use_multiprocessing`, `max_queue_size`. Do not pass these arguments to `fit()`, as they will be ignored.\n  self._warn_if_super_not_called()\n","output_type":"stream"},{"name":"stdout","text":"Epoch 1/15\n","output_type":"stream"},{"name":"stderr","text":"WARNING: All log messages before absl::InitializeLog() is called are written to STDERR\nI0000 00:00:1779120174.705122     136 service.cc:152] XLA service 0x79e2e8007c30 initialized for platform CUDA (this does not guarantee that XLA will be used). Devices:\nI0000 00:00:1779120174.705157     136 service.cc:160]   StreamExecutor device (0): Tesla T4, Compute Capability 7.5\nI0000 00:00:1779120174.705161     136 service.cc:160]   StreamExecutor device (1): Tesla T4, Compute Capability 7.5\nI0000 00:00:1779120182.495989     136 cuda_dnn.cc:529] Loaded cuDNN version 91002\n2026-05-18 16:03:12.404562: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:03:12.542136: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:03:13.992294: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:03:14.132409: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:03:15.122283: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:03:15.263342: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:03:17.575211: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:03:17.714430: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:03:18.069813: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:03:18.210901: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:03:20.402711: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:03:20.542274: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\nI0000 00:00:1779120224.222160     136 device_compiler.h:188] Compiled cluster using XLA!  This line is logged at most once for the lifetime of the process.\n","output_type":"stream"},{"name":"stdout","text":"\u001b[1m603/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m━\u001b[0m \u001b[1m9s\u001b[0m 1s/step - accuracy: 0.1300 - loss: 4.7292 ","output_type":"stream"},{"name":"stderr","text":"2026-05-18 16:14:05.073101: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:14:05.207199: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:14:06.606718: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:14:06.747940: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:14:07.705235: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:14:07.839911: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:14:09.586352: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:14:09.723535: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:14:11.293711: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:14:11.428279: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n","output_type":"stream"},{"name":"stdout","text":"\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 1s/step - accuracy: 0.1315 - loss: 4.7157","output_type":"stream"},{"name":"stderr","text":"/usr/local/lib/python3.12/dist-packages/keras/src/trainers/data_adapters/py_dataset_adapter.py:121: UserWarning: Your `PyDataset` class should call `super().__init__(**kwargs)` in its constructor. `**kwargs` can include `workers`, `use_multiprocessing`, `max_queue_size`. Do not pass these arguments to `fit()`, as they will be ignored.\n  self._warn_if_super_not_called()\n2026-05-18 16:17:25.643831: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:17:25.780809: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:17:27.126732: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n2026-05-18 16:17:27.267883: E external/local_xla/xla/stream_executor/cuda/cuda_timer.cc:86] Delay kernel timed out: measured time has sub-optimal accuracy. There may be a missing warmup execution, please investigate in Nsight Systems.\n","output_type":"stream"},{"name":"stdout","text":"\nEpoch 1: val_accuracy improved from -inf to 0.24285, saving model to /kaggle/working/best_bird_model.keras\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m915s\u001b[0m 1s/step - accuracy: 0.1317 - loss: 4.7142 - val_accuracy: 0.2428 - val_loss: 3.7225 - learning_rate: 0.0010\nEpoch 2/15\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 812ms/step - accuracy: 0.4450 - loss: 2.4749\nEpoch 2: val_accuracy did not improve from 0.24285\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m618s\u001b[0m 1s/step - accuracy: 0.4450 - loss: 2.4747 - val_accuracy: 0.1008 - val_loss: 4.5223 - learning_rate: 0.0010\nEpoch 3/15\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 793ms/step - accuracy: 0.5476 - loss: 1.9212\nEpoch 3: val_accuracy improved from 0.24285 to 0.28250, saving model to /kaggle/working/best_bird_model.keras\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m606s\u001b[0m 990ms/step - accuracy: 0.5477 - loss: 1.9212 - val_accuracy: 0.2825 - val_loss: 3.3326 - learning_rate: 0.0010\nEpoch 4/15\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 793ms/step - accuracy: 0.6180 - loss: 1.5575\nEpoch 4: val_accuracy improved from 0.28250 to 0.47159, saving model to /kaggle/working/best_bird_model.keras\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m628s\u001b[0m 1s/step - accuracy: 0.6179 - loss: 1.5575 - val_accuracy: 0.4716 - val_loss: 2.4071 - learning_rate: 0.0010\nEpoch 5/15\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 790ms/step - accuracy: 0.6639 - loss: 1.3778\nEpoch 5: val_accuracy did not improve from 0.47159\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m602s\u001b[0m 984ms/step - accuracy: 0.6638 - loss: 1.3779 - val_accuracy: 0.1944 - val_loss: 3.8994 - learning_rate: 0.0010\nEpoch 6/15\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 781ms/step - accuracy: 0.7084 - loss: 1.1734\nEpoch 6: val_accuracy did not improve from 0.47159\n\nEpoch 6: ReduceLROnPlateau reducing learning rate to 0.0005000000237487257.\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m599s\u001b[0m 979ms/step - accuracy: 0.7084 - loss: 1.1735 - val_accuracy: 0.4358 - val_loss: 2.6085 - learning_rate: 0.0010\nEpoch 7/15\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 797ms/step - accuracy: 0.7846 - loss: 0.8166\nEpoch 7: val_accuracy improved from 0.47159 to 0.70155, saving model to /kaggle/working/best_bird_model.keras\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m608s\u001b[0m 993ms/step - accuracy: 0.7846 - loss: 0.8166 - val_accuracy: 0.7016 - val_loss: 1.3857 - learning_rate: 5.0000e-04\nEpoch 8/15\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m0s\u001b[0m 803ms/step - accuracy: 0.8443 - loss: 0.5802\nEpoch 8: val_accuracy did not improve from 0.70155\n\u001b[1m612/612\u001b[0m \u001b[32m━━━━━━━━━━━━━━━━━━━━\u001b[0m\u001b[37m\u001b[0m \u001b[1m611s\u001b[0m 999ms/step - accuracy: 0.8443 - loss: 0.5802 - val_accuracy: 0.4980 - val_loss: 2.4799 - learning_rate: 5.0000e-04\nEpoch 9/15\n\u001b[1m 77/612\u001b[0m \u001b[32m━━\u001b[0m\u001b[37m━━━━━━━━━━━━━━━━━━\u001b[0m \u001b[1m6:55\u001b[0m 777ms/step - accuracy: 0.8903 - loss: 0.4016","output_type":"stream"}],"execution_count":null}]}