{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11558331,"sourceType":"datasetVersion","datasetId":7242517},{"sourceId":762,"sourceType":"modelInstanceVersion","modelInstanceId":629,"modelId":52},{"sourceId":356425,"sourceType":"modelInstanceVersion","modelInstanceId":297166,"modelId":317778}],"dockerImageVersionId":31011,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This notebook is inspired by the code from: https://medium.com/@danya.kosmin/meowtalk-how-to-train-yamnet-audio-classification-model-for-mobile-devices-4f228cf5650c\n\nAdditionally, see https://www.tensorflow.org/tutorials/audio/transfer_learning_audio#import_tensorflow_and_other_libraries","metadata":{}},{"cell_type":"markdown","source":"# Import libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport csv\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow_hub as hub\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport librosa\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, precision_recall_curve, average_precision_score\nfrom tensorflow.keras import layers, models","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T15:32:14.962922Z","iopub.execute_input":"2025-05-06T15:32:14.963159Z","iopub.status.idle":"2025-05-06T15:32:31.276840Z","shell.execute_reply.started":"2025-05-06T15:32:14.963135Z","shell.execute_reply":"2025-05-06T15:32:31.276315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set random seeds for reproducibility\nSEED = 42\ntf.random.set_seed(SEED)\nnp.random.seed(SEED)\n\nBATCH_SIZE = 32\nEPOCHS = 20\nINPUT_SHAPE = (1024,)  # YAMNet embedding size\nSAMPLE_RATE = 16000  # YAMNet requires 16kHz audio\n\nDATA_DIR = '/kaggle/input/birdclef-2025/train_audio'\nOUTPUT_DIR = '/kaggle/working/'\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T15:32:31.277528Z","iopub.execute_input":"2025-05-06T15:32:31.277929Z","iopub.status.idle":"2025-05-06T15:32:31.282623Z","shell.execute_reply.started":"2025-05-06T15:32:31.277907Z","shell.execute_reply":"2025-05-06T15:32:31.281946Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Handling Audio","metadata":{}},{"cell_type":"code","source":"def load_audio_file(file_path, sample_rate=16000):\n    audio, file_sr = librosa.load(file_path, sr=sample_rate, mono=True)\n    audio = audio.astype(np.float32)\n    return audio\n\ndef segment_audio(audio, sample_rate=16000):\n    segment_length = int(5 * sample_rate) # 5 seconds * sample rate = number of samples for the segment\n    segments = []\n    \n    # Pad audio if needed\n    if len(audio) % segment_length != 0:\n        padding = np.zeros(segment_length - (len(audio) % segment_length), dtype=np.float32)\n        audio = np.concatenate([audio, padding])\n    \n    # Split into segments\n    for i in range(0, len(audio), segment_length):\n        segment = audio[i:i+segment_length]\n        # Only include segments that have actual audio (not just silence)\n        if np.abs(segment).max() > 0.01:  # Really small threshold\n            segments.append(segment)\n    \n    return segments\n\ndef extract_yamnet_embeddings(audio_data, yamnet_model):\n    scores, embeddings, spectrogram  = yamnet_model(audio_data)\n    \n    # YamNet returns a result per 0.96s (basically 1s) of audio\n    # As we are working with 5s windows, we have to take the mean of the results\n    return tf.reduce_mean(embeddings, axis=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T15:32:31.284132Z","iopub.execute_input":"2025-05-06T15:32:31.284463Z","iopub.status.idle":"2025-05-06T15:32:31.301469Z","shell.execute_reply.started":"2025-05-06T15:32:31.284440Z","shell.execute_reply":"2025-05-06T15:32:31.300828Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def prepare_dataset(data_dir):\n    all_files = []\n    species_sets = []  # List of sets containing species in each recording\n    \n    species_folders = [d for d in os.listdir(data_dir) if os.path.isdir(os.path.join(data_dir, d))]\n    \n    for species in species_folders:\n        species_path = os.path.join(data_dir, species)\n        audio_files = [os.path.join(species_path, f) for f in os.listdir(species_path) \n                       if f.endswith('.ogg')]\n        \n        all_files.extend(audio_files)\n        species_sets.extend([{species}] * len(audio_files))\n    \n    # Create dataframe\n    df = pd.DataFrame({\n        'file_path': all_files,\n        'species_set': species_sets\n    })\n    \n    # Get all unique species\n    all_species = set()\n    for species_set in df['species_set']:\n        all_species.update(species_set)\n    all_species = sorted(list(all_species))\n    \n    # Create mapping dictionaries\n    species_to_idx = {species: i for i, species in enumerate(all_species)}\n    idx_to_species = {i: species for i, species in enumerate(all_species)}\n    \n    # Save mapping for later use\n    with open(os.path.join(OUTPUT_DIR, 'species_mapping.csv'), 'w', newline='') as f:\n        writer = csv.writer(f)\n        writer.writerow(['species', 'index'])\n        for species, idx in species_to_idx.items():\n            writer.writerow([species, idx])\n    \n    return df, species_to_idx, idx_to_species # Returns dataframe with file paths and labels.\n\n# Extract YAMNet embeddings for 5-second segments of all audio files, and make multi-hot encoded labels\ndef create_embeddings_dataset(df, yamnet_model, species_to_idx):\n\n    all_embeddings = []\n    all_labels = []\n    segment_file_mapping = []  # Track which file each segment came from\n    \n    for i, row in tqdm(df.iterrows(), total=len(df), desc=\"Extracting embeddings\"):\n        try:\n            # Load audio\n            audio = load_audio_file(row['file_path'])\n            \n            # Segment audio into 5-second chunks\n            segments = segment_audio(audio, SAMPLE_RATE)\n            \n            # Skip files with no valid segments\n            if not segments:\n                continue\n                \n            # Process each segment\n            for segment in segments:\n                # Extract embedding\n                embedding = extract_yamnet_embeddings(segment, yamnet_model)\n                all_embeddings.append(embedding.numpy())\n                \n                # Create multi-hot encoded label vector (for multi-label classification)\n                label_vector = np.zeros(len(species_to_idx))\n                for species in row['species_set']:\n                    label_vector[species_to_idx[species]] = 1\n                all_labels.append(label_vector)\n                \n                # Track which file this segment came from\n                segment_file_mapping.append(row['file_path'])\n        \n        except Exception as e:\n            tqdm.write(f\"Error processing {row['file_path']}: {e}\")\n            continue\n    \n    # Convert to numpy arrays\n    embeddings_array = np.array(all_embeddings)\n    labels_array = np.array(all_labels)\n    \n    # Create dataframe with segment info\n    segments_df = pd.DataFrame({\n        'file_path': segment_file_mapping\n    })\n    \n    # Save embeddings and labels to disk\n    np.save(os.path.join(OUTPUT_DIR, 'embeddings.npy'), embeddings_array)\n    np.save(os.path.join(OUTPUT_DIR, 'labels.npy'), labels_array)\n    segments_df.to_csv(os.path.join(OUTPUT_DIR, 'segments.csv'), index=False)\n    \n    return embeddings_array, labels_array, segments_df # Returns embeddings and multi-hot encoded labels.","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T15:32:31.302159Z","iopub.execute_input":"2025-05-06T15:32:31.302432Z","iopub.status.idle":"2025-05-06T15:32:31.314624Z","shell.execute_reply.started":"2025-05-06T15:32:31.302407Z","shell.execute_reply":"2025-05-06T15:32:31.314005Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Model (Helper) Function","metadata":{}},{"cell_type":"code","source":"# Added FocalLossBCE\ndef focal_loss_bce(gamma=2.0, alpha=0.25):\n    def focal_loss(y_true, y_pred):\n        epsilon = tf.keras.backend.epsilon()\n        y_pred = tf.clip_by_value(y_pred, epsilon, 1.0 - epsilon)\n        \n        # Binary cross entropy calculation\n        bce = tf.keras.backend.binary_crossentropy(y_true, y_pred)\n        \n        # Focal loss calculation\n        p_t = (y_true * y_pred) + ((1 - y_true) * (1 - y_pred))\n        alpha_factor = y_true * alpha + (1 - y_true) * (1 - alpha)\n        modulating_factor = tf.pow((1.0 - p_t), gamma)\n        \n        return tf.keras.backend.sum(alpha_factor * modulating_factor * bce, axis=-1)\n    \n    return focal_loss\n\ndef build_classifier_model(num_classes):\n    model = models.Sequential([\n        layers.Input(shape=INPUT_SHAPE), # Input from YamNet\n        layers.Dense(512, activation='relu'),\n        layers.Dropout(0.3),\n        layers.Dense(256, activation='relu'),\n        layers.Dropout(0.3),\n        layers.Dense(num_classes, activation='sigmoid')  # Sigmoid for probability per class\n    ])\n    \n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n        loss=tf.keras.losses.CategoricalFocalCrossentropy(\n            alpha=0.25,\n            gamma=2.0,\n            from_logits=False,\n            label_smoothing=0.0,\n            axis=-1,\n            reduction='sum_over_batch_size',\n            name='categorical_focal_crossentropy'\n        ),\n        metrics=['accuracy', tf.keras.metrics.AUC()]  # Include AUC metric\n    )\n    \n    return model\n\n\ndef train_and_evaluate_model(X_train, y_train, X_val, y_val, num_classes):\n    model = build_classifier_model(num_classes)\n    \n    # Create callbacks\n    early_stopping = tf.keras.callbacks.EarlyStopping(\n        monitor='val_loss',\n        patience=5,\n        restore_best_weights=True\n    )\n    \n    model_checkpoint = tf.keras.callbacks.ModelCheckpoint(\n        filepath=os.path.join(OUTPUT_DIR, 'best_model.keras'),\n        monitor='val_auc',\n        save_best_only=True\n    )\n    \n    # Train model\n    history = model.fit(\n        X_train, y_train,\n        validation_data=(X_val, y_val),\n        epochs=EPOCHS,\n        batch_size=BATCH_SIZE,\n        callbacks=[early_stopping, model_checkpoint]\n    )\n    \n    # Evaluate model\n    val_loss, val_accuracy, val_auc = model.evaluate(X_val, y_val)\n    print(f\"Validation loss: {val_loss:.4f}\")\n    print(f\"Validation accuracy: {val_accuracy:.4f}\")\n    print(f\"Validation AUC: {val_auc:.4f}\")\n   \n    return model\n\ndef evaluate_model(model, X_val, y_val, idx_to_species):\n    y_pred = model.predict(X_val)\n    \n    # Calculate per-class AUC\n    n_classes = y_val.shape[1]\n    auc_scores = []\n    \n    for i in range(n_classes):\n        # Only calculate AUC if there are positive examples\n        if np.sum(y_val[:, i]) > 0:\n            class_auc = roc_auc_score(y_val[:, i], y_pred[:, i])\n            auc_scores.append((idx_to_species[i], class_auc))\n            \n            # Plot precision-recall curve for this class\n            precision, recall, _ = precision_recall_curve(y_val[:, i], y_pred[:, i])\n            average_precision = average_precision_score(y_val[:, i], y_pred[:, i])\n    \n    # Calculate macro average AUC (the competition metric)\n    macro_auc = np.mean([auc for _, auc in auc_scores])\n    print(f\"Macro-averaged ROC-AUC: {macro_auc:.4f}\")\n    \n    # Save detailed AUC scores\n    auc_scores.sort(key=lambda x: x[1], reverse=True)\n    with open(os.path.join(OUTPUT_DIR, 'auc_scores.csv'), 'w', newline='') as f:\n        writer = csv.writer(f)\n        writer.writerow(['Species', 'AUC'])\n        for species, auc in auc_scores:\n            writer.writerow([species, auc])\n    \n    return macro_auc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T15:32:31.315253Z","iopub.execute_input":"2025-05-06T15:32:31.315478Z","iopub.status.idle":"2025-05-06T15:32:31.331060Z","shell.execute_reply.started":"2025-05-06T15:32:31.315456Z","shell.execute_reply":"2025-05-06T15:32:31.330387Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Prediction functions","metadata":{}},{"cell_type":"code","source":"# Create inference model that processes audio in 5-second windows\ndef create_inference_model(yamnet_model, classifier_model):\n    \n    class BirdSegmentClassifier(tf.keras.Model):\n        def __init__(self, yamnet_model, classifier_model):\n            super(BirdSegmentClassifier, self).__init__()\n            self.yamnet_model = yamnet_model\n            self.classifier_model = classifier_model\n            self.sample_rate = 16000\n            self.window_size = 5\n        \n        def process_audio(self, audio):\n            # Segment the audio into 5-second chunks\n            segments = segment_audio(audio, self.sample_rate)\n            \n            all_predictions = []\n            \n            for segment in segments:\n                _, embeddings, _ = self.yamnet_model(segment)\n                embedding = tf.reduce_mean(embeddings, axis=0)[tf.newaxis, :]\n                \n                # Get predictions from our classifier\n                predictions = self.classifier_model(embedding)\n                all_predictions.append(predictions[0])\n            \n            # If we have segments, average predictions across all segments\n            if all_predictions:\n                return tf.stack(all_predictions)\n            else:\n                # Return zeros if no valid segments\n                return tf.zeros((1, self.classifier_model.output_shape[-1]))\n        \n        def call(self, inputs):\n            return self.process_audio(inputs)\n    \n    inference_model = BirdSegmentClassifier(yamnet_model, classifier_model)\n    return inference_model\n    \n# Function to prepare submission format\ndef prepare_submission(test_audio_dir, inference_model, idx_to_species):\n    all_results = []\n    \n    # Process each audio file in the test directory\n    for audio_file in tqdm(os.listdir(test_audio_dir), desc=\"Processing test files\"):\n        if audio_file.endswith(('.ogg')):\n            file_path = os.path.join(test_audio_dir, audio_file)\n            \n            try:\n                # Load audio\n                audio = load_audio_file(file_path)\n                \n                # Process audio in 5-second segments\n                segments = segment_audio(audio, SAMPLE_RATE)\n                \n                # For each segment, predict species probabilities\n                for i, segment in enumerate(segments):\n                    # Extract embedding and predict\n                    _, embeddings, _ = inference_model.yamnet_model(segment)\n                    embedding = tf.reduce_mean(embeddings, axis=0)[tf.newaxis, :]\n                    predictions = inference_model.classifier_model(embedding)[0].numpy()\n                    \n                    # Create row_id (filename_segment)\n                    row_id = f\"{os.path.splitext(audio_file)[0]}_{i}\"\n                    \n                    # Store results\n                    result = {'row_id': row_id}\n                    for class_idx, prob in enumerate(predictions):\n                        species = idx_to_species[class_idx]\n                        result[species] = float(prob)\n                    \n                    all_results.append(result)\n            \n            except Exception as e:\n                print(f\"Error processing {file_path}: {e}\")\n                continue\n    \n    # Create submission dataframe\n    submission_df = pd.DataFrame(all_results)\n    \n    # Set row_id as index\n    submission_df.set_index('row_id', inplace=True)\n    \n    return submission_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T15:32:31.331890Z","iopub.execute_input":"2025-05-06T15:32:31.332524Z","iopub.status.idle":"2025-05-06T15:32:31.347335Z","shell.execute_reply.started":"2025-05-06T15:32:31.332506Z","shell.execute_reply":"2025-05-06T15:32:31.346834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Loading YamNet\\n\")\nyamnet_model = hub.load('/kaggle/input/yamnet/tensorflow2/yamnet/1')\n\n# Check if we already have pre-computed embeddings, this is only for ensuring the submission/prediction\n# is of the correct format.\n# In the inference notebook we'll only do loading of these files and the model\nif os.path.exists(\"/kaggle/input/yamnet-training-files-needed/embeddings.npy\"):\n    embedding_file = \"/kaggle/input/yamnet-training-files-needed/embeddings.npy\"\nelse:\n    embedding_file = os.path.join(OUTPUT_DIR, 'embeddings.npy')\n\nif os.path.exists(\"/kaggle/input/yamnet-training-files-needed/labels.npy\"):\n    labels_file = \"/kaggle/input/yamnet-training-files-needed/labels.npy\"\nelse:\n    labels_file = os.path.join(OUTPUT_DIR, 'labels.npy')\n\nif os.path.exists(\"/kaggle/input/yamnet-training-files-needed/segments.csv\"):\n    segments_file = \"/kaggle/input/yamnet-training-files-needed/segments.csv\"\nelse:\n    segments_file = os.path.join(OUTPUT_DIR, 'segments.csv')\n\nif os.path.exists(embedding_file) and os.path.exists(labels_file) and os.path.exists(segments_file):\n    print(\"Loading pre-computed embeddings\\n\")\n    embeddings = np.load(embedding_file)\n    labels = np.load(labels_file)\n    segments_df = pd.read_csv(segments_file)\n    \n    # Load species mappings\n    species_mapping = pd.read_csv(\"/kaggle/input/yamnet-training-files-needed/species_mapping.csv\")\n    species_to_idx = dict(zip(species_mapping['species'], species_mapping['index']))\n    idx_to_species = dict(zip(species_mapping['index'], species_mapping['species']))\nelse:\n    print(\"Preparing dataset\\n\")\n    df, species_to_idx, idx_to_species = prepare_dataset(DATA_DIR)\n    \n    print(f\"Total number of audio files: {len(df)}\")\n    print(f\"Number of species: {len(species_to_idx)}\")\n    \n    print(\"Computing embeddings for all audio segments\\n\")\n    embeddings, labels, segments_df = create_embeddings_dataset(df, yamnet_model, species_to_idx)\n\nprint(f\"Total embeddings: {len(embeddings)}\")\n\n# Split into train/validation sets\nindices = np.arange(len(embeddings))\ntrain_indices, val_indices = train_test_split(\n    indices, test_size=0.2, random_state=SEED, \n    stratify=np.argmax(labels, axis=1) if labels.shape[1] > 1 else labels\n)\n\nX_train = embeddings[train_indices]\ny_train = labels[train_indices]\nX_val = embeddings[val_indices]\ny_val = labels[val_indices]\n\nprint(f\"Training data: {len(X_train)} segments\")\nprint(f\"Validation data: {len(X_val)} segments\")\n\n\nprint(\"Training model/Loading model\\n\")\n# if os.path.exists(\"/kaggle/input/yamnetclassification-head/tensorflow2/default/1/best_model.keras\"):\n#     classifier_model = models.load_model(\"/kaggle/input/yamnetclassification-head/tensorflow2/default/1/best_model.keras\")\n#     print(\"Model loaded successfully.\")\n# else:\nprint(\"Training model\\n\")\nclassifier_model = train_and_evaluate_model(X_train, y_train, X_val, y_val, len(species_to_idx))\n\n\nprint(\"Evaluating model on validation set\\n\")\nmacro_auc = evaluate_model(classifier_model, X_val, y_val, idx_to_species)\nprint(f\"Competition metric (Macro-averaged ROC-AUC): {macro_auc:.4f}\")\n\n\nprint(\"Creating inference model\")\ninference_model = create_inference_model(yamnet_model, classifier_model)\n\n\n# For actual submission:\ntest_dir = '/kaggle/input/birdclef-2025/train_soundscapes'\nsubmission = prepare_submission(test_dir, inference_model, idx_to_species)\nsubmission.to_csv(os.path.join(OUTPUT_DIR, 'submission.csv'))\n\nprint(\"Models and evaluation results saved to: /kaggle/working/\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T15:33:27.487562Z","iopub.execute_input":"2025-05-06T15:33:27.487838Z","iopub.status.idle":"2025-05-06T16:04:07.272544Z","shell.execute_reply.started":"2025-05-06T15:33:27.487816Z","shell.execute_reply":"2025-05-06T16:04:07.271813Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T15:33:24.938208Z","iopub.status.idle":"2025-05-06T15:33:24.938520Z","shell.execute_reply.started":"2025-05-06T15:33:24.938376Z","shell.execute_reply":"2025-05-06T15:33:24.938394Z"}},"outputs":[],"execution_count":null}]}