{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport tensorflow as tf\nfrom tensorflow.keras import layers, models, mixed_precision\nfrom tqdm import tqdm\nfrom tensorflow.keras.applications import VGG19","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-26T13:44:42.26431Z","iopub.execute_input":"2025-04-26T13:44:42.264555Z","iopub.status.idle":"2025-04-26T13:44:56.82507Z","shell.execute_reply.started":"2025-04-26T13:44:42.26453Z","shell.execute_reply":"2025-04-26T13:44:56.824087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"strategy = tf.distribute.MirroredStrategy()\nmixed_precision.set_global_policy('mixed_float16')\nprint(f\"Number of GPUs: {strategy.num_replicas_in_sync}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T13:44:56.825942Z","iopub.execute_input":"2025-04-26T13:44:56.826371Z","iopub.status.idle":"2025-04-26T13:44:57.935352Z","shell.execute_reply.started":"2025-04-26T13:44:56.82635Z","shell.execute_reply":"2025-04-26T13:44:57.934393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"SAMPLE_RATE = 32000\nDURATION = 5\nN_MELS = 128\nIMG_SIZE = (128, 128)\nBATCH_SIZE = 32\nEPOCHS = 10","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T13:44:57.936262Z","iopub.execute_input":"2025-04-26T13:44:57.936609Z","iopub.status.idle":"2025-04-26T13:44:57.940252Z","shell.execute_reply.started":"2025-04-26T13:44:57.936578Z","shell.execute_reply":"2025-04-26T13:44:57.939646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_audio(path_tensor):\n    # Wrap librosa logic with tf.py_function to handle TensorFlow tensors\n    def _load_audio(path_bytes):\n        # Convert TensorFlow tensor (bytes) to Python string\n        path_str = \"/kaggle/input/birdclef-2025/train_audio/\" + path_bytes.numpy().decode('utf-8')\n        \n        # Load audio with librosa\n        audio, _ = librosa.load(\n            path_str,\n            sr=SAMPLE_RATE,\n            duration=DURATION,\n            res_type='kaiser_fast'\n        )\n        \n        # Pad/trim to exact duration\n        if len(audio) < SAMPLE_RATE * DURATION:\n            audio = np.pad(audio, (0, SAMPLE_RATE * DURATION - len(audio)))\n        else:\n            audio = audio[:SAMPLE_RATE * DURATION]\n        \n        # Create mel spectrogram\n        spectrogram = librosa.feature.melspectrogram(\n            y=audio, \n            sr=SAMPLE_RATE, \n            n_mels=N_MELS,\n            hop_length=512\n        )\n        spectrogram = librosa.power_to_db(spectrogram).astype(np.float32)\n        \n        # Add channel dimension and resize\n        spectrogram = np.expand_dims(spectrogram, axis=-1)\n        spectrogram = tf.image.resize(spectrogram, [128, 128])  # Shape: (128, 128, 1)\n        \n        # Standardization\n        mean = tf.math.reduce_mean(spectrogram)\n        std = tf.math.reduce_std(spectrogram)\n        spectrogram = (spectrogram - mean) / (std + 1e-6)\n        \n        return spectrogram\n    \n    # Execute the Python function as a TensorFlow op\n    spectrogram = tf.py_function(\n        _load_audio,\n        [path_tensor],\n        Tout=tf.float32  # Output type\n    )\n    \n    # Set fixed shape for TensorFlow's static shape inference\n    spectrogram.set_shape((128, 128, 1))\n    \n    return spectrogram\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T13:44:57.941898Z","iopub.execute_input":"2025-04-26T13:44:57.942098Z","iopub.status.idle":"2025-04-26T13:44:57.955193Z","shell.execute_reply.started":"2025-04-26T13:44:57.94208Z","shell.execute_reply":"2025-04-26T13:44:57.954538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_model(num_classes):\n    with strategy.scope():\n        inputs = layers.Input(shape=(128, 128, 1))\n        \n        x = layers.Concatenate(axis=-1)([inputs] * 3)\n        \n        # Load pre-trained DenseNet169 without top layers\n        base_model = VGG19(\n            include_top=False,\n            weights='imagenet',\n            input_shape=(128, 128, 3),\n            pooling=None\n        )\n        \n        # Freeze base model\n        base_model.trainable = False\n        \n        # Forward pass through DenseNet\n        x = base_model(x)\n        \n        # Add custom head\n        x = layers.GlobalAveragePooling2D()(x)\n        x = layers.Dense(512, activation='relu')(x)\n        x = layers.Dropout(0.5)(x)\n        outputs = layers.Dense(num_classes, activation='softmax', dtype='float32')(x)\n        \n        # Create and compile model\n        model = models.Model(inputs, outputs)\n        model.compile(\n            optimizer=tf.keras.optimizers.Adam(learning_rate=1e-4),\n            loss='categorical_crossentropy',\n            metrics=['categorical_accuracy']\n        )\n        return model\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T13:44:57.956829Z","iopub.execute_input":"2025-04-26T13:44:57.957019Z","iopub.status.idle":"2025-04-26T13:44:57.972311Z","shell.execute_reply.started":"2025-04-26T13:44:57.957003Z","shell.execute_reply":"2025-04-26T13:44:57.971534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_clip(clip):\n    # Implement identical preprocessing to training\n    # Including padding, spectrogram generation, normalization\n    # Return shape: (IMG_SIZE[0], IMG_SIZE[1], 1)\n    \n    if len(clip) < SAMPLE_RATE * 5:\n        clip = np.pad(clip, (0, SAMPLE_RATE * 5 - len(clip)))\n    \n    spec = librosa.feature.melspectrogram(\n        y=clip, \n        sr=SAMPLE_RATE, \n        n_mels=N_MELS\n    )\n    spec = librosa.power_to_db(spec)\n    spec = tf.image.resize(spec[..., np.newaxis], IMG_SIZE)\n    return spec.numpy().squeeze()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T13:44:57.973226Z","iopub.execute_input":"2025-04-26T13:44:57.973589Z","iopub.status.idle":"2025-04-26T13:44:57.988753Z","shell.execute_reply.started":"2025-04-26T13:44:57.973568Z","shell.execute_reply":"2025-04-26T13:44:57.987966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prepare_dataset(batch_size=32, validation_split=0.1):\n    # Load metadata\n    train_df = pd.read_csv('/kaggle/input/birdclef-2025/train.csv')\n    taxonomy = pd.read_csv('/kaggle/input/birdclef-2025/taxonomy.csv')\n\n    # Create full dataset\n    full_dataset = load_and_preprocess_data(train_df, taxonomy)\n\n    # Calculate dataset statistics\n    num_total_samples = len(train_df)\n    num_classes = len(taxonomy['primary_label'].unique())\n\n    # Split dataset\n    num_train = int(num_total_samples * (1 - validation_split))\n    num_val = num_total_samples - num_train\n    \n    # Create batched datasets\n    train_data = full_dataset.take(num_train).shuffle(1000)\n    train_data = train_data.batch(batch_size).prefetch(tf.data.AUTOTUNE)\n    \n    val_data = full_dataset.skip(num_train)\n    val_data = val_data.batch(batch_size).prefetch(tf.data.AUTOTUNE)\n    \n    return train_data, val_data, num_classes","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T13:44:57.989439Z","iopub.execute_input":"2025-04-26T13:44:57.989934Z","iopub.status.idle":"2025-04-26T13:44:58.000743Z","shell.execute_reply.started":"2025-04-26T13:44:57.989914Z","shell.execute_reply":"2025-04-26T13:44:57.999947Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Install required audio processing backend\n!pip install soundfile audioread","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T13:44:58.001653Z","iopub.execute_input":"2025-04-26T13:44:58.00192Z","iopub.status.idle":"2025-04-26T13:45:02.292296Z","shell.execute_reply.started":"2025-04-26T13:44:58.001894Z","shell.execute_reply":"2025-04-26T13:45:02.291265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_label(row, label_to_idx):\n    label_name = row[1]\n    label_idx = label_to_idx[label_name]\n    return tf.one_hot(label_idx, depth=num_classes)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T13:45:02.293901Z","iopub.execute_input":"2025-04-26T13:45:02.29416Z","iopub.status.idle":"2025-04-26T13:45:02.298485Z","shell.execute_reply.started":"2025-04-26T13:45:02.294138Z","shell.execute_reply":"2025-04-26T13:45:02.297543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_and_preprocess_data(train_df, taxonomy):\n    # Implement your actual data loading and preprocessing here\n    # Should return tf.data.Dataset containing (spectrogram, label) pairs\n    # Example structure:\n    file_paths = train_df['filename'].tolist()\n    # Process labels\n    valid_labels = set(taxonomy['primary_label'])\n    label_to_idx = {label: idx for idx, label in enumerate(valid_labels)}\n    print(len(valid_labels))\n    # Create dataset\n    dataset = tf.data.Dataset.from_generator(\n        lambda: ((process_audio(fp), create_label(row, label_to_idx)) \n                for fp, row in zip(file_paths, train_df.itertuples())),\n        output_signature=(\n            tf.TensorSpec(shape=(128, 128, 1)),\n            tf.TensorSpec(shape=(len(valid_labels)),)\n        )\n    )\n    \n    return dataset.cache().prefetch(tf.data.AUTOTUNE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T13:45:02.29946Z","iopub.execute_input":"2025-04-26T13:45:02.299857Z","iopub.status.idle":"2025-04-26T13:45:02.334089Z","shell.execute_reply.started":"2025-04-26T13:45:02.299826Z","shell.execute_reply":"2025-04-26T13:45:02.333243Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    full_dataset = prepare_dataset()\n\n    # Calculate dataset size using original file count (pre-batching)\n    taxonomy = pd.read_csv('/kaggle/input/birdclef-2025/taxonomy.csv')\n    train_df = pd.read_csv('/kaggle/input/birdclef-2025/train.csv')\n    num_samples = len(train_df)  # Original dataset size\n    \n    train_data = full_dataset[0]\n    val_data = full_dataset[1]\n\n    # Create model\n    num_classes = len(taxonomy['primary_label'].unique())\n    model = create_model(num_classes)\n\n    # Train model\n    history = model.fit(\n        train_data,\n        validation_data=val_data,\n        epochs=EPOCHS,\n        steps_per_epoch=800,\n        callbacks=[\n            tf.keras.callbacks.ModelCheckpoint('best_model.keras', \n                                             save_best_only=True,\n                                             save_weights_only=False),\n            tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss',\n                                               patience=2,\n                                               factor=0.5,\n                                               verbose=1),\n            tf.keras.callbacks.TensorBoard(log_dir='./logs', profile_batch='500,520')\n        ]\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T13:45:02.334935Z","iopub.execute_input":"2025-04-26T13:45:02.335214Z","iopub.status.idle":"2025-04-26T14:20:05.360333Z","shell.execute_reply.started":"2025-04-26T13:45:02.335187Z","shell.execute_reply":"2025-04-26T14:20:05.3594Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"    taxonomy = pd.read_csv('/kaggle/input/birdclef-2025/taxonomy.csv')\n    species_ids = taxonomy['primary_label'].tolist()\n    submission = pd.read_csv('/kaggle/input/birdclef-2025/sample_submission.csv', index_col='row_id')\n    \n    model.load_weights('best_model.keras')\n    \n    test_files = [f for f in os.listdir('test_soundscapes') if f.endswith('.ogg')]\n    for file_name in tqdm(test_files):\n        soundscape_id = file_name.split('_')[1].split('.')[0]\n        audio, _ = librosa.load(\n            os.path.join('test_soundscapes', file_name),\n            sr=SAMPLE_RATE\n        )\n        \n        for start_time in range(0, 60, 5):\n            end_time = start_time + 5\n            row_id = f'soundscape_{soundscape_id}_{end_time}'\n            if row_id not in submission.index:\n                continue\n\n            clip = audio[start_time*SAMPLE_RATE:end_time*SAMPLE_RATE]\n            if len(clip) < SAMPLE_RATE * 5:\n                clip = np.pad(clip, (0, SAMPLE_RATE*5 - len(clip)))\n            \n            spec = librosa.feature.melspectrogram(\n                y=clip, sr=SAMPLE_RATE, n_mels=N_MELS)\n            spec = librosa.power_to_db(spec)\n            spec = tf.image.resize(spec[..., np.newaxis], IMG_SIZE).numpy()\n            \n            preds = model.predict(np.expand_dims(spec, 0), verbose=0)[0]\n            submission.loc[row_id, species_ids] = preds\n    \n    submission.reset_index().to_csv('submission.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-26T14:20:05.361355Z","iopub.execute_input":"2025-04-26T14:20:05.361891Z","iopub.status.idle":"2025-04-26T14:20:08.077768Z","shell.execute_reply.started":"2025-04-26T14:20:05.361867Z","shell.execute_reply":"2025-04-26T14:20:08.076062Z"}},"outputs":[],"execution_count":null}]}