{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":417835,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":340833,"modelId":362080}],"dockerImageVersionId":31040,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport gc\nimport warnings\nimport logging\nimport time\nimport math\nimport cv2\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport timm\nfrom timm import create_model\nfrom tqdm.auto import tqdm\n\n# Suppress warnings and limit logging output\nwarnings.filterwarnings(\"ignore\")\nlogging.basicConfig(level=logging.ERROR)\n\"\"\"\nsave all the params and config class used in the inference pipeline\n\"\"\"\ntest_soundscapes = '/kaggle/input/birdclef-2025/test_soundscapes'  \nsubmission_csv = '/kaggle/input/birdclef-2025/sample_submission.csv'  \ntaxonomy_csv = '/kaggle/input/birdclef-2025/taxonomy.csv' \nmodel_path = '/kaggle/input/linear/pytorch/default/1' \n\n# test_soundscapes = '/mnt/sda/lht/Kaggle/test_soundscapes'  \n# submission_csv = '/mnt/sda/lht/Kaggle/sample_submission.csv'  \n# taxonomy_csv = '/mnt/sda/lht/Kaggle/taxonomy.csv' \n# model_path = '/mnt/sda/lht/Kaggle/linear' \n    \n\n# sound params\nFS = 32000  # sampling frequency(32kHz)\nWINDOW_SIZE = 5  # cutting window size (s)\n    \n# Mel\nN_FFT = 1024  # FFT window size\nHOP_LENGTH = 256\nN_MELS = 256 # filter number\nFMIN = 50  # minmum frequency #20\nFMAX = 14000  # max freq　#16000\nTARGET_SHAPE = (256, 256)  # Mel image size\n    \n# model params\nmodel_name = 'regnety_008'  # model name\nin_channels = 1  # input channel\ndevice = 'cpu'\n    \n# 推論パラメータ\nuse_tta = False  # test time augmentation\ntta_count = 3  # aug time\nthreshold = 0.7  \n    \nuse_specific_folds = False  \nfolds = [0, 1] \n    \n# debug setting\ndebug = False  \ndebug_count = 5  \n\nif debug:\n    test_soundscapes = '/kaggle/input/birdclef-2025/train_soundscapes'\n\nclass BirdCLEF2025Pipeline:\n\n    class BirdCLEFModel(nn.Module):\n        def __init__(self, feat_dim, num_classes):\n            super().__init__()\n            # self.encoder = encoder\n            self.encoder = create_model(\"regnety_008\", pretrained=False, num_classes=0)\n            # for param in self.encoder.parameters():\n            #     param.requires_grad = False\n            self.gap = nn.AdaptiveAvgPool2d((1, 1))\n            self.classifier = nn.Linear(feat_dim, num_classes)\n    \n        def forward(self, x):\n            # with torch.no_grad():\n            #     feat = self.encoder.forward_features(x)\n            # print(x.shape)\n            # feat = self.encoder(x)\n            feat = self.encoder.forward_features(x)\n            # feat = self.gap(feat).view(x.size(0), -1)\n            # return self.classifier(feat)\n            feat = self.gap(feat).view(x.size(0), -1)\n            return self.classifier(feat)\n\n    # class BirdCLEFModel(nn.Module):\n    #     def __init__(self, num_classes):\n    #         super().__init__()\n    #         # self.cfg = cfg\n    #         self.backbone = timm.create_model(\n    #             model_name, # regnety_008\n    #             pretrained=False,  \n    #             in_chans=in_channels,   # 1 \n    #             drop_rate=0.0,    \n    #             drop_path_rate=0.0\n    #         )\n\n    #         if 'efficientnet' in model_name:\n    #             backbone_out = self.backbone.classifier.in_features # channel\n    #             self.backbone.classifier = nn.Identity()  # remove classifier\n    #         elif 'resnet' in model_name:\n    #             backbone_out = self.backbone.fc.in_features         # channel\n    #             self.backbone.fc = nn.Identity()          # remove classifier\n    #         else:\n    #             backbone_out = self.backbone.get_classifier().in_features\n    #             self.backbone.reset_classifier(0, '')\n            \n    #         self.pooling = nn.AdaptiveAvgPool2d(1)  \n    #         self.feat_dim = backbone_out # backbone nerwork output dim\n    #         self.classifier = nn.Linear(backbone_out, num_classes)  # classification head\n            \n    #     def forward(self, x):\n    #         \"\"\"\n    #         input: [bs 1, h, w]\n    #         output: [bs, num_classes] (logits)\n    #         \"\"\"\n    #         features = self.backbone(x)   # feature extraction\n    #         if isinstance(features, dict):\n    #             features = features['features']\n            \n    #         if len(features.shape) == 4: # [batch, ch, W, H]\n    #             features = self.pooling(features)\n    #             features = features.view(features.size(0), -1) # [batch, ch]\n    #         logits = self.classifier(features)  # [baatch, num_classes]\n    #         return logits\n\n    def __init__(self):\n        \"\"\"\n        load \n        \"\"\"\n        # self.cfg = cfg\n        self.taxonomy_df = None\n        self.species_ids = []\n        self.models = []\n        self._load_taxonomy()  \n\n    def _load_taxonomy(self):\n        \"\"\"\n        load class info from taxonmy.csv\n        self.species_ids: class id\n        \"\"\"\n        print(\"loading class info...\")\n        self.taxonomy_df = pd.read_csv(taxonomy_csv)\n        self.species_ids = self.taxonomy_df['primary_label'].tolist() \n        print(f\"number of classes: {len(self.species_ids)}\")\n\n    def audio2melspec(self, audio_data):\n        \"\"\"\n        audio file to mel sepc(image with 1 channel)\n        \"\"\"\n        if np.isnan(audio_data).any():\n            mean_signal = np.nanmean(audio_data)\n            audio_data = np.nan_to_num(audio_data, nan=mean_signal)\n        \n        mel_spec = librosa.feature.melspectrogram(\n            y=audio_data,\n            sr=FS,\n            n_fft=N_FFT,\n            hop_length=HOP_LENGTH,\n            n_mels=N_MELS,\n            fmin=FMIN,\n            fmax=FMAX,\n            power=2.0\n        )\n        mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n        mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n        return mel_spec_norm\n\n    def process_audio_segment(self, audio_data):\n        \n        # short then padding\n        if len(audio_data) < FS * WINDOW_SIZE:\n            audio_data = np.pad(\n                audio_data,\n                (0, FS * WINDOW_SIZE - len(audio_data)),\n                mode='constant'\n            )\n        \n        mel_spec = self.audio2melspec(audio_data)  # メルスペクトログラムに変換\n        \n        # reshape to fit model input shape\n        if mel_spec.shape != TARGET_SHAPE:  # (256, 256)\n            mel_spec = cv2.resize(mel_spec, TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n            \n        return mel_spec.astype(np.float32)\n\n    def find_model_files(self):\n        model_files = []\n        model_dir = Path(model_path)\n        for path in model_dir.glob('**/*.pth'):\n            model_files.append(str(path))\n        return model_files\n\n    def load_models(self):\n        self.models = []\n        model_files = self.find_model_files() \n        if not model_files:\n            print(\"model file not found!\") \n            return self.models\n\n        print(f\" total {len(model_files)} models\")\n        \n        if use_specific_folds:\n            filtered_files = []\n            for fold in folds:\n                fold_files = [f for f in model_files if f\"fold{fold}\" in f]\n                filtered_files.extend(fold_files)\n            model_files = filtered_files\n            print(f\"specific file({folds}) for {len(model_files)}will utilize single model file\")\n        \n        for model_path in model_files:\n            try:\n                print(f\"loading model: {model_path}\")\n                checkpoint = torch.load(model_path, map_location=torch.device(device), weights_only = False)\n                model = self.BirdCLEFModel(768, len(self.species_ids))\n                state_dict = checkpoint.get(\"state_dict\", checkpoint)\n                model.load_state_dict(state_dict)\n                # state_dict = torch.load(path)\n                # model.load_state_dict(state_dict)\n                model = model.to(device)\n                model.eval()  # eval mode\n                self.models.append(model)\n            except Exception as e:\n                print(f\"loading model {model_path} an error occurs: {e}\")\n        \n        return self.models  # a list of models for ensemble inference??\n\n    def apply_tta(self, spec, tta_idx):\n        \"\"\"\n        augmentation on mel spec\n        \"\"\"\n        if tta_idx == 0:\n            # no aug\n            return spec\n        elif tta_idx == 1:\n            # horizontal flip\n            return np.flip(spec, axis=1)\n        elif tta_idx == 2:\n            # vertical flip\n            return np.flip(spec, axis=0)\n        else:\n            return spec\n\n    def predict_on_spectrogram(self, audio_path):\n        predictions = []\n        row_ids = []\n        soundscape_id = Path(audio_path).stem\n        \n        try:\n            print(f\"{soundscape_id}processing...\")\n            audio_data, _ = librosa.load(audio_path, sr=FS)\n            total_segments = int(len(audio_data) / (FS * WINDOW_SIZE))  # 5s a fragment\n            \n            for segment_idx in range(total_segments):\n                start_sample = segment_idx * FS * WINDOW_SIZE\n                end_sample = start_sample + FS * WINDOW_SIZE\n                segment_audio = audio_data[start_sample:end_sample]\n                \n                end_time_sec = (segment_idx + 1) * WINDOW_SIZE\n                row_id = f\"{soundscape_id}_{end_time_sec}\"\n                row_ids.append(row_id)\n\n                if use_tta:\n                    all_preds = []\n                    for tta_idx in range(tta_count):\n                        mel_spec = self.process_audio_segment(segment_audio)\n                        mel_spec = self.apply_tta(mel_spec, tta_idx)\n                        # mel_spec_tensor = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n                        mel_spec_tensor = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0)\n                        mel_spec_tensor = mel_spec_tensor.repeat(3, 1, 1).unsqueeze(0)\n                        mel_spec_tensor = mel_spec_tensor.to(device)\n                        print(f\"mel_spec_tensor shape: {mel_spec_tensor.shape}\")\n\n\n                        if len(self.models) == 1:\n                            with torch.no_grad():\n                                outputs = self.models[0](mel_spec_tensor)   # forward processs for current fragment\n                                probs = torch.sigmoid(outputs).cpu().numpy().squeeze()  # l\n                                all_preds.append(probs)\n                        else:\n                            segment_preds = []\n                            for model in self.models:\n                                with torch.no_grad():\n                                    outputs = model(mel_spec_tensor)\n                                    probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                                    segment_preds.append(probs)\n                            avg_preds = np.mean(segment_preds, axis=0)\n                            all_preds.append(avg_preds)\n                    final_preds = np.mean(all_preds, axis=0)\n                else:\n                    mel_spec = self.process_audio_segment(segment_audio)\n                    # mel_spec_tensor = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0).unsqueeze(0)\n                    # mel_spec_tensor = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0)\n                    # mel_spec_tensor = mel_spec_tensor.repeat(3, 1, 1).unsqueeze(0)\n                    mel_spec_tensor = torch.tensor(mel_spec, dtype=torch.float32).unsqueeze(0)\n                    mel_spec_tensor = mel_spec_tensor.expand(3, -1, -1).unsqueeze(0) \n                    mel_spec_tensor = mel_spec_tensor.to(device)\n                    # print(f\"mel_spec_tensor shape: {mel_spec_tensor.shape}\")\n                    if len(self.models) == 1:\n                        with torch.no_grad():\n                            outputs = self.models[0](mel_spec_tensor)\n                            \n                            final_preds = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                    else:\n                        segment_preds = []\n                        for model in self.models:\n                            with torch.no_grad():\n                                outputs = model(mel_spec_tensor)\n                                probs = torch.sigmoid(outputs).cpu().numpy().squeeze()\n                                segment_preds.append(probs)\n                        final_preds = np.mean(segment_preds, axis=0)  # ensemble prediction??\n                \n                predictions.append(final_preds)\n        except Exception as e:\n            print(f\"processing {audio_path}an error occurs: {e}\")\n        \n        return row_ids, predictions\n\n    def run_inference(self):\n        \"\"\"\n        uisng method predict_on_spectrogram to classify each audio file\n        :return: all_row_ids & corresponding predictions\n        \"\"\"\n        test_files = list(Path(test_soundscapes).glob('*.ogg'))  \n        if debug:\n            print(f\"activate debug mode. Utilizing {debug_count} examples\")\n            test_files = test_files[:debug_count]\n        print(f\"{len(test_files)} number of test files is founded\")\n\n        all_row_ids = []\n        all_predictions = []\n\n        for audio_path in tqdm(test_files):\n            row_ids, predictions = self.predict_on_spectrogram(str(audio_path))\n            all_row_ids.extend(row_ids)\n            all_predictions.extend(predictions)\n        \n        return all_row_ids, all_predictions\n\n    def create_submission(self, row_ids, predictions):\n\n        # print(\"提出用データフレームを作成中...\")\n        submission_dict = {'row_id': row_ids}\n        for i, species in enumerate(self.species_ids):  # classes\n            submission_dict[species] = [pred[i] for pred in predictions]\n\n        submission_df = pd.DataFrame(submission_dict)\n        submission_df.set_index('row_id', inplace=True)\n\n        sample_sub = pd.read_csv(submission_csv, index_col='row_id')\n        missing_cols = set(sample_sub.columns) - set(submission_df.columns)\n        if missing_cols:\n            print(f\"{len(missing_cols)} kind of missing data in the results\")\n            for col in missing_cols:\n                submission_df[col] = 0.0\n\n        submission_df = submission_df[sample_sub.columns] \n        submission_df = submission_df.reset_index()\n        \n        return submission_df\n\n    def smooth_submission(self, submission_path):\n        \n        # print(\"提出結果の予測を平滑化しています...\")\n        sub = pd.read_csv(submission_path)\n        cols = sub.columns[1:]\n        # 'row_id'を基にグループを抽出\n        groups = sub['row_id'].str.rsplit('_', n=1).str[0].values\n        unique_groups = np.unique(groups)\n        \n        for group in unique_groups:\n            idx = np.where(groups == group)[0]\n            sub_group = sub.iloc[idx].copy()\n            predictions = sub_group[cols].values\n            new_predictions = predictions.copy()\n            \n            if predictions.shape[0] > 1:\n                new_predictions[0] = (predictions[0] * 0.8) + (predictions[1] * 0.2)\n                new_predictions[-1] = (predictions[-1] * 0.8) + (predictions[-2] * 0.2)\n                for i in range(1, predictions.shape[0]-1):\n                    new_predictions[i] = (predictions[i-1] * 0.2) + (predictions[i] * 0.6) + (predictions[i+1] * 0.2)\n            sub.iloc[idx, 1:] = new_predictions\n        \n        sub.to_csv(submission_path, index=False)\n        print(f\"smoothed result is saved to {submission_path}\")\n\n    def run(self):\n\n        start_time = time.time()\n        print(\"BirdCLEF-2025 start inference...\")\n        # print(f\"TTA有効: {self.cfg.use_tta} (変動数: {self.cfg.tta_count if self.cfg.use_tta else 0})\")\n    \n        self.load_models()\n        if not self.models:\n            print(\"no model founded\")\n            return\n    \n        print(f\"use model numbers: {len(self.models)}\")\n        row_ids, predictions = self.run_inference()\n        submission_df = self.create_submission(row_ids, predictions)\n    \n        submission_path = 'submission.csv'\n        submission_df.to_csv(submission_path, index=False)\n        print(f\"submission saved at {submission_path}\")\n    \n        self.smooth_submission(submission_path)\n    \n        end_time = time.time()\n        print(f\"inference finished (time needed {(end_time - start_time) / 60:.2f} minute)\")\n\n\n# cfg = CFG()\nprint(f\"Using device: {device}\")\npipeline = BirdCLEF2025Pipeline()\npipeline.run()  # Use the correct method name here\n","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}