{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":11261159,"sourceType":"datasetVersion","datasetId":7038322},{"sourceId":11273427,"sourceType":"datasetVersion","datasetId":7047426},{"sourceId":11390008,"sourceType":"datasetVersion","datasetId":7132849},{"sourceId":11477557,"sourceType":"datasetVersion","datasetId":7193543},{"sourceId":11477562,"sourceType":"datasetVersion","datasetId":7193548},{"sourceId":11498252,"sourceType":"datasetVersion","datasetId":7208180},{"sourceId":11516482,"sourceType":"datasetVersion","datasetId":7222268},{"sourceId":11561307,"sourceType":"datasetVersion","datasetId":7249056},{"sourceId":11746360,"sourceType":"datasetVersion","datasetId":7373899}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport time\nimport warnings\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset\nfrom pathlib import Path\nfrom tqdm.auto import tqdm\nimport cv2\nimport timm\n\nwarnings.filterwarnings('ignore')\n\n# Configuration\nclass CFG:\n    # Yollar - Kaggle ortamı için\n    test_soundscapes = \"/kaggle/input/birdclef-2025/test_soundscapes\"\n    sample_submission = \"/kaggle/input/birdclef-2025/sample_submission.csv\"\n    taxonomy_csv = \"/kaggle/input/birdclef-2025/taxonomy.csv\"\n    \n    # Eğitilmiş modellerin yolu\n    kaggle_model_path = \"/kaggle/input/seed-2023-numwork2/seed-2023-fold-0.pth\"\n    local_model_2 = \"/kaggle/input/final-model-fold-01/final_model_fold1.pth\"\n    local_model_1 = \"/kaggle/input/0-788-imp-seed-77-fold-0/0.788-imp-seed-77-fold-0.pth\"\n    \n    # Ses işleme parametreleri\n    sr = 32000\n    duration = 5\n    \n    # Spectogram parametreleri\n    n_mels = 224\n    n_fft = 2048\n    hop_length = 256\n    fmin = 20\n    fmax = 16000\n    img_size = 224\n    \n    # Tahmin parametreleri - Hızlandırılmış\n    batch_size = 64        # Daha büyük batch boyutu\n    use_tta = True          # TTA kullanımı korundu (performans için)\n    tta_count = 3           # TTA sayısı\n    \n    # Donanım\n    device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    \n    # Hata ayıklama\n    debug = False\n    debug_count = 5\n\n# Önceden hesaplayabileceğimiz değişkenler\nSEGMENT_SAMPLES = CFG.duration * CFG.sr\n\n# Yardımcı işlevler - Hafif optimizasyonlar\ndef mono_to_color(X, eps=1e-6, mean=None, std=None):\n    \"\"\"Grayscale spektrogramı normalize et ve renkli görüntüye dönüştür\"\"\"\n    mean = mean or X.mean()\n    std = std or X.std()\n    X = (X - mean) / (std + eps)\n    \n    # Clipleme ve [0, 255] aralığına normalize etme\n    _min, _max = X.min(), X.max()\n    if (_max - _min) > eps:\n        V = np.clip(X, _min, _max)\n        V = 255 * (V - _min) / (_max - _min)\n        V = V.astype(np.uint8)\n    else:\n        V = np.zeros_like(X, dtype=np.uint8)\n    \n    # RGB kanallarını oluştur\n    return np.stack([V, V, V], axis=-1)\n\ndef audio_to_image(audio, sr=32000, n_mels=224, n_fft=2048, hop_length=256, \n                   fmin=20, fmax=16000, img_size=224):\n    \"\"\"Ses verisini spektrogram görüntüsüne dönüştür\"\"\"\n    # Mel spektrogramı oluştur\n    melspec = librosa.feature.melspectrogram(\n        y=audio, \n        sr=sr, \n        n_mels=n_mels,\n        n_fft=n_fft, \n        hop_length=hop_length, \n        fmin=fmin,\n        fmax=fmax\n    )\n    \n    # dB cinsinden dönüştür\n    melspec = librosa.power_to_db(melspec, ref=np.max)\n    \n    # Renk dönüşümü ve boyutlandırma\n    image = mono_to_color(melspec)\n    image = cv2.resize(image, (img_size, img_size))\n    \n    # Kanalları modelin beklediği sıraya getir (HWC -> CHW)\n    image = np.moveaxis(image, -1, 0)\n    image = image.astype(np.float32) / 255.0\n    \n    return image\n\n# Test-time augmentation\ndef apply_tta(image, tta_idx):\n    # Keep only the 3 most effective TTAs\n    if tta_idx == 0:\n        return image\n    elif tta_idx == 1:\n        return torch.flip(image, dims=[-1])\n    elif tta_idx == 2:\n        mask = torch.ones_like(image)\n        t = image.shape[-1] // 8\n        mask[:, :, t:2*t, :] = 0\n        return image * mask\n\n# Model sınıfı - Eğitim kodlarıyla aynı olmalı\nclass BirdCLEFModel(nn.Module):\n    def __init__(self, model_name, num_classes, in_channels=3):\n        super(BirdCLEFModel, self).__init__()\n        self.model_name = model_name\n        self.num_classes = num_classes\n        \n        # timm kütüphanesiyle model oluştur\n        self.backbone = timm.create_model(\n            model_name,\n            pretrained=False,\n            in_chans=in_channels\n        )\n        \n        # Özellik boyutunu al\n        if hasattr(self.backbone, 'classifier'):\n            in_features = self.backbone.classifier.in_features\n            self.backbone.classifier = nn.Identity()\n        elif hasattr(self.backbone, 'fc'):\n            in_features = self.backbone.fc.in_features\n            self.backbone.fc = nn.Identity()\n        else:\n            in_features = self.backbone.num_features if hasattr(self.backbone, 'num_features') else 1280\n        \n        # Özel sınıflandırıcı\n        self.classifier = nn.Sequential(\n            nn.Linear(in_features, 512),\n            nn.ReLU(inplace=True),\n            nn.Dropout(0.3),\n            nn.Linear(512, num_classes)\n        )\n    \n    def forward(self, x):\n        # Omurga özellikleri\n        features = self.backbone(x)\n        \n        # Özellik tensörünün boyutunu kontrol et ve gerekirse düzleştir\n        if len(features.shape) == 4:\n            features = F.adaptive_avg_pool2d(features, output_size=1).squeeze(-1).squeeze(-1)\n        \n        # Sınıflandırıcı\n        logits = self.classifier(features)\n        \n        return logits\n\n# Modelleri yükle - Optimizasyonlar\ndef load_models(cfg, species_ids):\n    \"\"\"Kaggle ve Yerel modelleri yükle\"\"\"\n    models = []\n    \n    # Kaggle'da eğitilmiş fold0 modelini yükle\n    if os.path.exists(cfg.kaggle_model_path):\n        print(f\"Kaggle'da eğitilmiş fold0 modeli yükleniyor: {cfg.kaggle_model_path}\")\n        \n        # Model oluştur\n        model_kaggle = BirdCLEFModel('efficientnet_b0', len(species_ids))\n        \n        # Ağırlıkları yükle\n        checkpoint = torch.load(cfg.kaggle_model_path, map_location=cfg.device)\n        model_kaggle.load_state_dict(checkpoint['model_state_dict'])\n        model_kaggle = model_kaggle.to(cfg.device)\n        model_kaggle.eval()  # Değerlendirme moduna getir\n        \n        print(f\"Kaggle fold0 modeli başarıyla yüklendi, AUC: {checkpoint.get('best_auc', 'N/A')}\")\n        models.append(model_kaggle)\n    else:\n        print(f\"UYARI: Kaggle fold0 modeli bulunamadı: {cfg.kaggle_model_path}\")\n\n    if os.path.exists(cfg.local_model_2):\n        print(f\"Yerel ortamda eğitilmiş model2 yükleniyor: {cfg.local_model_2}\")\n        \n        # Model oluştur\n        model_local_2 = BirdCLEFModel('efficientnet_b0', len(species_ids))\n        \n        # Ağırlıkları yükle\n        checkpoint = torch.load(cfg.local_model_2, map_location=cfg.device)\n        model_local_2.load_state_dict(checkpoint['model_state_dict'])\n        model_local_2 = model_local_2.to(cfg.device)\n        model_local_2.eval()  # Değerlendirme moduna getir\n        \n        print(f\"Yerel model2 başarıyla yüklendi, AUC: {checkpoint.get('best_auc', 'N/A')}\")\n        models.append(model_local_2)\n    else:\n        print(f\"UYARI: Yerel model2 bulunamadı: {cfg.local_model_2}\")\n\n\n\n\n    if os.path.exists(cfg.local_model_1):\n        print(f\"Yerel ortamda eğitilmiş model1 yükleniyor: {cfg.local_model_1}\")\n        \n        # Model oluştur\n        model_local_1 = BirdCLEFModel('efficientnet_b0', len(species_ids))\n        \n        # Ağırlıkları yükle\n        checkpoint = torch.load(cfg.local_model_1, map_location=cfg.device)\n        model_local_1.load_state_dict(checkpoint['model_state_dict'])\n        model_local_1 = model_local_1.to(cfg.device)\n        model_local_1.eval()  # Değerlendirme moduna getir\n        \n        print(f\"Yerel model1 başarıyla yüklendi, AUC: {checkpoint.get('best_auc', 'N/A')}\")\n        models.append(model_local_1)\n    else:\n        print(f\"UYARI: Yerel model1 bulunamadı: {cfg.local_model_1}\")\n\n\n    # En az bir model var mı kontrol et\n    if len(models) == 0:\n        raise ValueError(\"Hiçbir model yüklenemedi! Lütfen model yollarını kontrol edin.\")\n    \n    print(f\"Toplam {len(models)} model başarıyla yüklendi.\")\n    return models\n\n# Ses üzerinde tahmin yapma - Batch işleme ile optimize edildi\ndef predict_on_audio(audio_path, models, cfg, species_ids):\n    \"\"\"Ses dosyası üzerinde tahmin yap\"\"\"\n    predictions = []\n    row_ids = []\n    soundscape_id = Path(audio_path).stem\n    \n    try:\n        print(f\"İşleniyor: {soundscape_id}\")\n        \n        # Ses dosyasını yükle\n        try:\n            audio_data, _ = librosa.load(audio_path, sr=cfg.sr)\n        except Exception as e:\n            print(f\"Ses dosyası okuma hatası. Güvenli modda yeniden deneniyor: {audio_path}\")\n            import soundfile as sf\n            audio_data, orig_sr = sf.read(audio_path)\n            if len(audio_data.shape) > 1:  # Stereo -> Mono\n                audio_data = librosa.to_mono(audio_data.T)\n            if orig_sr != cfg.sr:\n                audio_data = librosa.resample(audio_data, orig_sr=orig_sr, target_sr=cfg.sr)\n        \n        # Segmentlere böl (5 saniyelik pencereler)\n        total_segments = int(len(audio_data) / SEGMENT_SAMPLES)\n        \n        # Batch işleme için listeler\n        batch_images = []\n        batch_rows = []\n        batch_indices = []\n        \n        for segment_idx in range(total_segments):\n            start_sample = segment_idx * SEGMENT_SAMPLES\n            end_sample = start_sample + SEGMENT_SAMPLES\n            segment_audio = audio_data[start_sample:end_sample]\n            \n            # Son segmentin uzunluğunu kontrol et\n            if len(segment_audio) < SEGMENT_SAMPLES:\n                segment_audio = np.pad(segment_audio, (0, SEGMENT_SAMPLES - len(segment_audio)), 'constant')\n            \n            # Row ID oluştur (her 5 saniyelik segment için)\n            end_time_sec = (segment_idx + 1) * cfg.duration\n            row_id = f\"{soundscape_id}_{end_time_sec}\"\n            \n            # Ses segmentini görüntüye dönüştür\n            img = audio_to_image(\n                segment_audio,\n                sr=cfg.sr,\n                n_mels=cfg.n_mels,\n                n_fft=cfg.n_fft,\n                hop_length=cfg.hop_length,\n                fmin=cfg.fmin,\n                fmax=cfg.fmax,\n                img_size=cfg.img_size\n            )\n            \n            # Batch'e ekle\n            batch_images.append(img)\n            batch_rows.append(row_id)\n            batch_indices.append(segment_idx)\n            \n            # Batch doldu mu veya son segment mi kontrol et\n            if len(batch_images) >= cfg.batch_size or segment_idx == total_segments - 1:\n                # Batch'i işle\n                if batch_images:\n                    # Tüm modellerin tahminlerini sakla\n                    all_models_preds = []\n                    \n                    # Tensöre dönüştür\n                    batch_tensor = torch.tensor(np.array(batch_images), dtype=torch.float32)\n                    batch_tensor = batch_tensor.to(cfg.device)\n                    \n                    for model in models:\n                        # TTA kullanılacak mı?\n                        if cfg.use_tta:\n                            all_tta_preds = []\n                            \n                            for tta_idx in range(cfg.tta_count):\n                                # TTA uygula\n                                tta_tensor = apply_tta(batch_tensor, tta_idx)\n                                \n                                # Tahmin\n                                with torch.no_grad():\n                                    outputs = model(tta_tensor)\n                                    outputs = torch.sigmoid(outputs).cpu().numpy()\n                                    all_tta_preds.append(outputs)\n                            \n                            # TTA tahminlerinin ortalaması\n                            model_preds = np.mean(all_tta_preds, axis=0)\n                        else:\n                            # TTA kullanmadan tahmin\n                            with torch.no_grad():\n                                outputs = model(batch_tensor)\n                                model_preds = torch.sigmoid(outputs).cpu().numpy()\n                        \n                        all_models_preds.append(model_preds)\n                    \n                    # Tüm modellerin tahminlerini ortala (ensemble)\n                    # Modellerin ağırlıkları\n                    if len(models) >= 3:\n                        batch_ensemble_preds = all_models_preds[0] * 0.2 + all_models_preds[1] * 0.5 + all_models_preds[2] * 0.3\n                    elif len(models) == 2:\n                        batch_ensemble_preds = all_models_preds[0] * 0.25 + all_models_preds[1] * 0.75\n                    else:\n                        batch_ensemble_preds = all_models_preds[0]\n                    \n                    # Sonuçları ekle\n                    row_ids.extend(batch_rows)\n                    predictions.extend(batch_ensemble_preds)\n                    \n                    # Batch'i temizle\n                    batch_images = []\n                    batch_rows = []\n                    batch_indices = []\n            \n    except Exception as e:\n        print(f\"Hata: {audio_path} dosyasını işlerken hata oluştu: {e}\")\n    \n    return row_ids, predictions\n\n# Tahminleri düzleştir (zaman içinde yumuşatma)\ndef smooth_predictions(row_ids, predictions, win_size=5):\n    \"\"\"Tahminleri zamansal olarak düzleştir\"\"\"\n    if len(predictions) <= 1:\n        return predictions\n    \n    # Aynı ses dosyasına ait segmentleri grupla\n    row_id_parts = [r.rsplit('_', 1)[0] for r in row_ids]\n    unique_groups = np.unique(row_id_parts)\n    \n    smoothed_predictions = np.copy(predictions)\n    \n    for group in unique_groups:\n        # Bu gruba ait tahminleri bul\n        group_indices = [i for i, r in enumerate(row_id_parts) if r == group]\n        group_preds = predictions[group_indices]\n        \n        # Her segment için kaydırma penceresi uygula\n        for i in range(len(group_indices)):\n            # Pencere sınırlarını belirle\n            win_start = max(0, i - win_size // 2)\n            win_end = min(len(group_indices), i + win_size // 2 + 1)\n            \n            # Pencere içindeki tahminlerin ağırlıklı ortalaması\n            if win_end - win_start > 1:\n                # Merkezdeki tahmine daha fazla ağırlık ver\n                weights = np.ones(win_end - win_start)\n                center_idx = i - win_start\n                if 0 <= center_idx < len(weights):\n                    weights[center_idx] = 2.0  # Merkeze daha fazla ağırlık\n                weights = weights / weights.sum()\n                \n                weighted_sum = np.zeros_like(group_preds[0])\n                for w_idx, pred_idx in enumerate(range(win_start, win_end)):\n                    weighted_sum += weights[w_idx] * group_preds[pred_idx]\n                \n                smoothed_predictions[group_indices[i]] = weighted_sum\n    \n    return smoothed_predictions\n\n# Tahmin işlemi - Paralel işleme için optimize edildi\ndef run_inference(cfg, models, species_ids):\n    \"\"\"Tüm test dosyaları üzerinde tahmin yap\"\"\"\n    test_files = list(Path(cfg.test_soundscapes).glob('*.ogg'))\n    \n    if cfg.debug:\n        print(f\"Debug modu: Sadece {cfg.debug_count} dosya işlenecek\")\n        test_files = test_files[:cfg.debug_count]\n    \n    print(f\"{len(test_files)} test ses dosyası bulundu\")\n    \n    # Test dosyası yoksa\n    if len(test_files) == 0:\n        print(f\"Uyarı: {cfg.test_soundscapes} dizininde .ogg dosyası bulunamadı!\")\n        print(\"Bu muhtemelen yerel test ortamında beklenen bir durum.\")\n        print(\"Kaggle submission sırasında dosyalar otomatik olarak eklenecektir.\")\n        \n        # Örnek submission'dan blank template oluştur\n        sample_sub = pd.read_csv(cfg.sample_submission)\n        return sample_sub['row_id'].tolist(), [np.zeros(len(species_ids)) for _ in range(len(sample_sub))]\n    \n    all_row_ids = []\n    all_predictions = []\n    \n    for audio_path in tqdm(test_files):\n        row_ids, predictions = predict_on_audio(str(audio_path), models, cfg, species_ids)\n        \n        if row_ids and len(row_ids) > 0:\n            all_row_ids.extend(row_ids)\n            all_predictions.extend(predictions)\n    \n    # NumPy dizisine dönüştür\n    if all_predictions:\n        all_predictions = np.array(all_predictions)\n        \n        # Tahminleri düzleştir\n        all_predictions = smooth_predictions(all_row_ids, all_predictions, win_size=5)\n    \n    return all_row_ids, all_predictions\n\n# Submission oluştur\ndef create_submission(row_ids, predictions, species_ids, cfg):\n    \"\"\"Kaggle submission dosyası oluştur\"\"\"\n    print(\"Submission dosyası oluşturuluyor...\")\n    \n    # Submission sözlüğü oluştur\n    submission_dict = {'row_id': row_ids}\n    \n    # Her tür için tahminleri ekle\n    for i, species in enumerate(species_ids):\n        submission_dict[species] = [pred[i] for pred in predictions]\n    \n    # DataFrame oluştur\n    submission_df = pd.DataFrame(submission_dict)\n    \n    # Örnek submission ile karşılaştır ve eksik satırları kontrol et\n    sample_sub = pd.read_csv(cfg.sample_submission)\n    \n    missing_rows = set(sample_sub['row_id']) - set(submission_df['row_id'])\n    if missing_rows:\n        print(f\"Uyarı: {len(missing_rows)} satır eksik. Tahmin: {len(row_ids)}, Gerekli: {len(sample_sub)}\")\n        \n        # Eksik satırlar için sıfır tahmin ekle\n        for row_id in missing_rows:\n            missing_dict = {'row_id': row_id}\n            for species in species_ids:\n                missing_dict[species] = 0.0\n            \n            # DataFrame'e ekle\n            missing_df = pd.DataFrame([missing_dict])\n            submission_df = pd.concat([submission_df, missing_df], ignore_index=True)\n    \n    # row_id'ye göre sırala\n    submission_df = submission_df.sort_values('row_id').reset_index(drop=True)\n    \n    return submission_df\n\ndef main():\n    start_time = time.time()\n    print(\"BirdCLEF 2025 ensemble submission süreci başlatılıyor...\")\n    print(f\"Kaggle model: {CFG.kaggle_model_path}\")\n    print(f\"Yerel model 2: {CFG.local_model_2}\")\n    print(f\"Yerel model 1: {CFG.local_model_1}\")\n    \n    # Tür bilgilerini yükle\n    taxonomy_df = pd.read_csv(CFG.taxonomy_csv)\n    species_ids = taxonomy_df['primary_label'].tolist()\n    num_classes = len(species_ids)\n    print(f\"{num_classes} türü için tahmin yapılacak.\")\n    \n    # Modelleri yükle\n    models = load_models(CFG, species_ids)\n    \n    # Tahmin yap\n    print(\"Test ses dosyaları üzerinde tahmin yapılıyor...\")\n    row_ids, predictions = run_inference(CFG, models, species_ids)\n    \n    # Submission oluştur\n    submission_df = create_submission(row_ids, predictions, species_ids, CFG)\n    \n    # Submission dosyasını kaydet\n    submission_path = \"submission.csv\"\n    submission_df.to_csv(submission_path, index=False)\n    print(f\"Submission dosyası kaydedildi: {submission_path}\")\n    \n    # İstatistikler\n    print(\"\\nSubmission İstatistikleri:\")\n    print(f\"Toplam satır sayısı: {len(submission_df)}\")\n    print(f\"Tahmin edilen örnek sayısı: {len(row_ids)}\")\n    \n    # Tahmin dağılımı kontrolü\n    pred_means = submission_df.iloc[:, 1:].mean().mean()\n    pred_std = submission_df.iloc[:, 1:].std().mean()\n    print(f\"Ortalama tahmin değeri: {pred_means:.4f}\")\n    print(f\"Tahmin standart sapması: {pred_std:.4f}\")\n    \n    # Çalışma süresi\n    end_time = time.time()\n    minutes = (end_time - start_time) / 60\n    print(f\"Tamamlandı! Çalışma süresi: {minutes:.2f} dakika\")\n\nif __name__ == \"__main__\":\n    main()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-09T11:01:43.270166Z","iopub.execute_input":"2025-05-09T11:01:43.270529Z","iopub.status.idle":"2025-05-09T11:01:44.788978Z","shell.execute_reply.started":"2025-05-09T11:01:43.270505Z","shell.execute_reply":"2025-05-09T11:01:44.787908Z"}},"outputs":[],"execution_count":null}]}