{"metadata":{"kernelspec":{"display_name":"Python 3 (ipykernel)","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.12.4"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":15853,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":2739,"modelId":319}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#pip install opencv-python\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#pip install geopandas","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#pip install folium","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport seaborn as sns\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow_hub as hub\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.utils import Sequence\nfrom tensorflow.keras.applications.efficientnet import EfficientNetB3\nfrom tensorflow.keras.layers import Input, GlobalAveragePooling2D, Dense, Dropout\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.preprocessing.image  import ImageDataGenerator\nfrom sklearn.metrics import classification_report, accuracy_score, f1_score, precision_score, recall_score,confusion_matrix\nimport librosa\nimport random\nimport time\n\nimport cv2\nimport matplotlib.pyplot as plt\nfrom glob import glob\nimport geopandas as gpd\nfrom shapely.geometry import Point\nimport folium\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint\nfrom pathlib import Path","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5"},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === PATHS & DATA ===\nDATA_DIR = Path(\"C:\\\\Users\\\\Asus\\\\Desktop\\\\birdclef-2025\")\ntrain_audio = DATA_DIR / \"train_audio\"\ntrain_soundscapes = DATA_DIR / \"train_soundscapes\"\ntest_soundscapes = DATA_DIR / \"test_soundscapes\"\ntrain_df= pd.read_csv(DATA_DIR / \"train.csv\")\ntaxonomy_df = pd.read_csv(DATA_DIR / \"taxonomy.csv\")\nSAMPLE_SUB = pd.read_csv(DATA_DIR / \"sample_submission.csv\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\n# Map all available files with full paths\ndef index_all_audio_files(train_audio_path):\n    audio_index = {}\n    for f in Path(train_audio_path).rglob(\"*.ogg\"):\n        audio_index[f.name] = f\n    return audio_index\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"taxonomy_df.columns","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"taxonomy_df.info()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\ntrain_audio_dir = Path(\"/kaggle/input/birdclef-2025/train_audio\")\nogg_files = list(train_audio_dir.rglob(\"*.ogg\"))\n\nprint(f\"Total .ogg files found: {len(ogg_files)}\")\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count number of clips per species (primary label)\nspecies_counts = train_df['primary_label'].value_counts().reset_index()\nspecies_counts.columns = ['primary_label', 'num_clips']\nprint(species_counts.head())\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Merge with taxonomy to get class_name\nspecies_info = species_counts.merge(taxonomy_df[['primary_label', 'common_name', 'class_name']], on='primary_label')\nprint(species_info.head())\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"bird_df = species_info[species_info['class_name'] == 'Aves']\nprint(f\"Total bird species: {len(bird_df)}\")\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(12, 6))\nbird_df.sort_values('num_clips', ascending=False).head(30).plot(\n    x='common_name', y='num_clips', kind='bar', legend=False, color='skyblue')\nplt.ylabel(\"Number of Audio Clips\")\nplt.title(\"Top 30 Bird Species by Clip Count\")\nplt.xticks(rotation=75)\nplt.tight_layout()\nplt.show()\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"First 5 train_df filenames:\")\nprint(train_df['filename'].head().tolist())\n\nprint(\"\\nFirst 5 audio_index keys:\")\nraw_index = index_all_audio_files(train_audio)\naudio_index = {k.lower().strip(): v for k, v in raw_index.items()}\nprint(list(audio_index.keys())[:5])\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"valid_filenames = set(audio_index.keys())\nfiltered_df = train_df[train_df['filename'].isin(valid_filenames)]\nprint(f\"Filtered from {len(train_df)} to {len(filtered_df)}\")\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 2. Constants\nSAMPLE_RATE = 32000\nDURATION = 5  # in seconds\nNUM_CLASSES = len(taxonomy_df)  # total classes in BirdCLEF+ 2025\nINPUT_SHAPE = (300, 300, 3)\nCONFIDENCE_THRESHOLD = 0.6\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def audio_to_spectrogram(file_path):\n    file_path = Path(file_path)\n    \n    # Basic checks\n    if not file_path.exists():\n        raise FileNotFoundError(f\"File not found: {file_path}\")\n    if file_path.suffix.lower() != '.ogg':\n        raise ValueError(f\"Unsupported file type (not .ogg): {file_path}\")\n    \n    # Load audio\n    y, sr = librosa.load(file_path, sr=SAMPLE_RATE, duration=DURATION)\n    \n    # Audio integrity checks\n    if y is None or len(y) < sr * 1:\n        raise ValueError(f\"Audio too short or empty in {file_path}\")\n    if np.max(np.abs(y)) < 0.001:\n        raise ValueError(f\"Audio is mostly silent in {file_path}\")\n\n    # Generate spectrogram\n    melspec = librosa.feature.melspectrogram(y=y, sr=sr, n_mels=128)\n    logmel = librosa.power_to_db(melspec)\n    img = cv2.resize(logmel, (300, 300))\n    img = np.stack([img, img, img], axis=-1)\n    return img / 255.0\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === FILE INDEXING ===\ndef index_all_audio_files(audio_root):\n    audio_index = {}\n    for f in Path(audio_root).rglob(\"*.ogg\"):\n        audio_index[f.name.lower()] = f\n    print(f\"Indexed {len(audio_index)} audio files.\")\n    return audio_index","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AudioDataGenerator(Sequence):\n    def __init__(self, df, audio_index, taxonomy_df, batch_size=32, augment=False):\n        self.df = df.reset_index(drop=True)\n        self.audio_index = audio_index\n        self.taxonomy_df = taxonomy_df\n        self.batch_size = batch_size\n        self.augment = augment\n        print(f\"[✓] Initialized AudioDataGenerator with {len(self.df)} samples.\")\n\n    def __len__(self):\n        return int(np.ceil(len(self.df) / self.batch_size))\n\n    def __getitem__(self, idx):\n        batch = self.df.iloc[idx*self.batch_size:(idx+1)*self.batch_size]\n        X, y = [], []\n        for _, row in batch.iterrows():\n            try:\n                file_path = str(self.audio_index[row['filename']])\n                spec = audio_to_spectrogram(file_path)\n                if spec.shape != INPUT_SHAPE:\n                    raise ValueError(\"Incorrect spectrogram shape\")\n                if self.augment:\n                    spec = self.augment_spectrogram(spec)\n                X.append(spec)\n                label = np.zeros(NUM_CLASSES)\n                idx_tax = self.taxonomy_df[self.taxonomy_df['primary_label'] == row['primary_label']].index[0]\n                label[idx_tax] = 1\n                y.append(label)\n            except Exception as e:\n                print(f\"Error loading {file_path}: {e}\")\n                continue\n        print(f\"[✓] Generated batch {idx + 1}/{self.__len__()} with {len(X)} samples.\")\n        return np.array(X), np.array(y)\n\n    def augment_spectrogram(self, spec):\n        if random.random() < 0.5:\n            spec += np.random.normal(0, 0.01, spec.shape)\n        if random.random() < 0.5:\n            t0 = random.randint(0, spec.shape[1] - 10)\n            spec[:, t0:t0 + 10, :] = 0\n        if random.random() < 0.5:\n            f0 = random.randint(0, spec.shape[0] - 10)\n            spec[f0:f0 + 10, :, :] = 0\n        if random.random() < 0.5:\n            spec = np.roll(spec, random.randint(-20, 20), axis=1)\n        if random.random() < 0.5:\n            spec = np.roll(spec, random.randint(-5, 5), axis=0)\n        if random.random() < 0.3:\n            spec += np.random.normal(0, 0.03, spec.shape)  # Stronger noise\n\n        return np.clip(spec, 0, 1)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === LOAD TRAINING + VALIDATION GENERATORS ===\ndef load_training_generator(split=0.1):\n    train_df['filename'] = train_df['filename'].apply(lambda x: Path(x).name.lower().strip())\n    audio_index = {f.name.lower(): f for f in Path(train_audio).rglob(\"*.ogg\")}\n    filtered_df = train_df[train_df['filename'].isin(audio_index)].copy()\n    train_df_split, val_df_split = train_test_split(filtered_df, test_size=split, stratify=filtered_df['primary_label'], random_state=42)\n\n    return (\n        AudioDataGenerator(train_df_split, audio_index, taxonomy_df, augment=True),\n        AudioDataGenerator(val_df_split, audio_index, taxonomy_df, augment=False)\n    )","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === BUILD MODEL ===\ndef build_model():\n    inputs = Input(shape=INPUT_SHAPE)\n    base = EfficientNetB3(include_top=False, weights='imagenet', input_tensor=inputs)\n    base.trainable = False\n    x = GlobalAveragePooling2D()(base.output)\n    outputs = Dense(NUM_CLASSES, activation='sigmoid')(x)\n    return Model(inputs, outputs)\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === TRAIN MODEL (with callbacks and monitoring) ===\ndef train_model(model, train_gen, val_gen, X_pseudo, y_pseudo):\n    X_val, y_val = val_gen[0]\n    X_train, y_train = train_gen[0]\n    X_train = np.concatenate([X_train, X_pseudo])\n    y_train = np.concatenate([y_train, y_pseudo])\n\n    # Callbacks\n    callbacks = [\n        tf.keras.callbacks.EarlyStopping(patience=2, restore_best_weights=True),\n        tf.keras.callbacks.ModelCheckpoint(\"checkpoint_best_model.h5\", save_best_only=True),\n        tf.keras.callbacks.TensorBoard(log_dir=\"logs\")\n    ]\n\n    history = {}\n    print(\"Starting base training...\")\n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n    hist1 = model.fit(\n        X_train, y_train,\n        validation_data=(X_val, y_val),\n        epochs=20,\n        callbacks=callbacks\n    )\n    history['pretrain'] = hist1.history\n\n    print(\"Starting fine-tuning...\")\n    model.trainable = True\n    for layer in model.layers[:100]:  # freeze shallow layers\n        layer.trainable = False\n\n    model.compile(optimizer=tf.keras.optimizers.Adam(1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    hist2 = model.fit(\n        X_train, y_train,\n        validation_data=(X_val, y_val),\n        epochs=20,  # Let EarlyStopping decide\n        callbacks=callbacks\n    )\n    history['finetune'] = hist2.history\n\n\n    # Plot confusion matrix on validation data\n    y_val_pred = model.predict(X_val)\n    y_val_bin = (y_val_pred > 0.5).astype(int)\n    cm = confusion_matrix(np.argmax(y_val, axis=1), np.argmax(y_val_bin, axis=1))\n    plt.figure(figsize=(10, 8))\n    sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\")\n    plt.title(\"Validation Confusion Matrix\")\n    plt.xlabel(\"Predicted\")\n    plt.ylabel(\"True\")\n    plt.show()\n\n    return model, history\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def generate_pseudo_labels(audio_dir, model, taxonomy_df, threshold=CONFIDENCE_THRESHOLD):\n    start_time = time.time()\n    pseudo_X, pseudo_y = [], []\n    audio_files = list(Path(audio_dir).rglob(\"*.ogg\"))\n    print(f\"[✓] Found {len(audio_files)} audio files for pseudo-labeling.\")\n\n    for file in audio_files:\n        try:\n            spec = audio_to_spectrogram(str(file))\n            if spec.shape != INPUT_SHAPE:\n                print(f\"[✗] Skipping {file.name}, invalid spectrogram shape: {spec.shape}\")\n                continue\n            prob = model.predict(np.expand_dims(spec, axis=0), verbose=0)[0]\n            if np.max(prob) > threshold:\n                pseudo_X.append(spec)\n                pseudo_y.append(prob)\n                print(f\"[✓] Pseudo-label from {file.name} with max prob {np.max(prob):.2f}\")\n            else:\n                print(f\"[✗] Low confidence ({np.max(prob):.2f}) on {file.name}, skipped.\")\n        except Exception as e:\n            print(f\"[!] Pseudo-labeling error for {file.name}: {e}\")\n\n    print(f\"[✓] Total pseudo-labeled samples: {len(pseudo_X)} in {time.time() - start_time:.2f} seconds.\")\n    return np.array(pseudo_X), np.array(pseudo_y)\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === GEO FILTERING ===\ndef build_geo_map():\n    geo_map = {}\n    for _, row in train_df.iterrows():\n        geo_map[row['filename']] = (row['latitude'], row['longitude'])\n    print(f\"Built geo_map with {len(geo_map)} entries.\")\n    return geo_map\n\ndef apply_geo_filtering(preds, df, geo_map, threshold=0.5):\n    filtered_preds = preds.copy()\n    count = 0\n    for i, row in df.iterrows():\n        latlon = geo_map.get(row['filename'])\n        if latlon:\n            if latlon[0] < -10:  # dummy filtering condition\n                filtered_preds[i] = (preds[i] > threshold) * preds[i]\n                count += 1\n    print(f\"Applied geo filtering to {count} entries.\")\n    return filtered_preds","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import folium\n\n# === TRAINING HISTORY ===\ndef plot_training_history(history):\n    for phase in history:\n        plt.figure(figsize=(12, 4))\n        plt.subplot(1, 2, 1)\n        plt.plot(history[phase]['accuracy'], label='train')\n        plt.plot(history[phase]['val_accuracy'], label='val')\n        plt.title(f'{phase} accuracy')\n        plt.legend()\n        plt.subplot(1, 2, 2)\n        plt.plot(history[phase]['loss'], label='train')\n        plt.plot(history[phase]['val_loss'], label='val')\n        plt.title(f'{phase} loss')\n        plt.legend()\n        plt.tight_layout()\n        plt.savefig(f\"training_curve_{phase}.png\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === EVALUATION ===\ndef evaluate_predictions(y_true, y_pred):\n    y_pred_bin = (y_pred > 0.5).astype(int)\n    print(\"Accuracy:\", accuracy_score(y_true, y_pred_bin))\n    print(\"F1 (macro):\", f1_score(y_true, y_pred_bin, average='macro'))\n    print(\"F1 (micro):\", f1_score(y_true, y_pred_bin, average='micro'))\n    print(\"Precision:\", precision_score(y_true, y_pred_bin, average='macro'))\n    print(\"Recall:\", recall_score(y_true, y_pred_bin, average='macro'))","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === SUBMISSION ===\ndef create_submission(preds, row_ids, geo_filtered=False):\n    df = pd.DataFrame(preds, columns=[f\"c{i}\" for i in range(preds.shape[1])])\n    df.insert(0, \"row_id\", row_ids)\n    suffix = \"geo\" if geo_filtered else \"raw\"\n    df.to_csv(f\"submission_{suffix}.csv\", index=False)\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === TEST TIME AUGMENTATION (TTA) ===\ndef predict_with_tta(model, X, n=3):\n    print(f\"Running TTA with {n} augmentations...\")\n    preds = np.zeros((len(X), NUM_CLASSES))\n    for i in range(n):\n        X_aug = np.array([np.clip(x + np.random.normal(0, 0.01, x.shape), 0, 1) for x in X])\n        preds += model.predict(X_aug, verbose=0)\n        print(f\"TTA round {i + 1}/{n} completed.\")\n    return preds / n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# === MAP ===\ndef plot_predictions_on_map(preds, metadata, threshold=0.5):\n    fmap = folium.Map(location=[0, -60], zoom_start=3)\n    for i, row in metadata.iterrows():\n        lat, lon = row['latitude'], row['longitude']\n        if pd.notnull(lat) and pd.notnull(lon):\n            ids = np.where(preds[i] > threshold)[0]\n            if len(ids):\n                label = \", \".join([f\"Species {sid}\" for sid in ids])\n                folium.Marker(location=[lat, lon], popup=label).add_to(fmap)\n    fmap.save(\"prediction_map.html\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def run_pipeline():\n    print(\"Loading training data...\")\n    train_gen, val_gen = load_training_generator()\n\n    print(\"Building model...\")\n    model = build_model()\n\n    print(\"Generating pseudo-labels...\")\n    X_pseudo, y_pseudo = generate_pseudo_labels(\n        audio_dir=DATA_DIR / \"train_soundscapes\",\n        model=model,\n        taxonomy_df=taxonomy_df\n    )\n\n    print(\"Training model with pseudo-labeled data...\")\n    model, history = train_model(model, train_gen, val_gen, X_pseudo, y_pseudo)\n\n    model.save(\"efficientnetb3_finetuned.h5\")\n    print(\"✅ Saved model as efficientnetb3_finetuned.h5\")\n\n    print(\"Plotting training history...\")\n    plot_training_history(history)\n\n    print(\"Evaluating on validation set...\")\n    X_val, y_val = val_gen[0]\n    y_val_pred = model.predict(X_val)\n    evaluate_predictions(y_val, y_val_pred)\n\n    print(\"Predicting test soundscapes...\")\n    test_files = list((DATA_DIR / \"test_soundscapes\").glob(\"*.ogg\"))\n    X_test = []\n\n    for f in test_files:\n        try:\n            if not f.exists():\n                print(f\"[✗] File does not exist: {f}\")\n                continue\n            if f.suffix.lower() != \".ogg\":\n                print(f\"[✗] Skipping non-ogg file: {f}\")\n                continue\n            spec = audio_to_spectrogram(f)\n            if spec.shape != INPUT_SHAPE:\n                print(f\"[✗] Invalid spectrogram shape {spec.shape} for file: {f.name}\")\n                continue\n            X_test.append(spec)\n        except Exception as e:\n            print(f\"[✗] Error processing {f.name}: {e}\")\n\n    if len(X_test) == 0:\n        print(\"[✗] No valid test spectrograms were created. Please check your test files.\")\n        return\n\n    X_test = np.array(X_test)\n    row_ids = [f\"soundscape_{f.stem}_{i*5}\" for f in test_files for i in range(12)]\n\n    print(\"Running TTA predictions...\")\n    preds = predict_with_tta(model, X_test)\n    create_submission(preds, row_ids, geo_filtered=False)\n\n    print(\"Applying geo filtering...\")\n    geo_map = build_geo_map()\n    preds_geo = apply_geo_filtering(preds.copy(), train_df.iloc[:len(preds)], geo_map)\n    create_submission(preds_geo, row_ids, geo_filtered=True)\n\n    plot_predictions_on_map(preds_geo, train_df.iloc[:len(preds_geo)])\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Run\nif __name__ == \"__main__\":\n    run_pipeline()","metadata":{"scrolled":true},"outputs":[],"execution_count":null}]}