{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11606750,"sourceType":"datasetVersion","datasetId":7277400}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import and config","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n%matplotlib inline\nimport os, gc, random \nimport pandas as pd\nimport pickle\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\nimport IPython.display as ipd\nfrom IPython.display import display, clear_output\nimport ipywidgets as widgets\nimport librosa\nimport librosa.display\nimport soundfile as sf\nimport numpy as np\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, accuracy_score, confusion_matrix\n\nclear_output()\nprint(\"Install and Import DONE\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T05:03:32.031853Z","iopub.execute_input":"2025-04-30T05:03:32.032865Z","iopub.status.idle":"2025-04-30T05:03:32.043071Z","shell.execute_reply.started":"2025-04-30T05:03:32.032830Z","shell.execute_reply":"2025-04-30T05:03:32.042114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nclass Config:\n    def __init__(self, **kwargs):\n        for k, v in kwargs.items():\n            setattr(self, k, v)\n\n    def update(self, **kwargs):\n        for k, v in kwargs.items():\n            setattr(self, k, v)\n\n# Initialize and set basic configuration\ncfg = Config(\n    DATA_FOLD_IDX = 0, # 0,1,2,3 \n    SEED=42, \n    SAMPLE_RATE=32000,\n    MEATA_DATA_PATH = Path('/kaggle/input/birdclef-2025/train.csv'),\n    SUBMISSION_SAMPLE_PATH = Path('/kaggle/input/birdclef-2025/sample_submission.csv'),\n    DATA_PATH=Path(\"/kaggle/input/birdclef-2025/train_audio\"),\n    OUTPUT_FOLDER =Path(\"/kaggle/working/\"),\n    MODEL_PATH = Path(\"/kaggle/input/bc25-models-pt-files/latest_model.pt\"),\n    COLOR_MAP ='inferno',\n    NUM_CLASSES = 206,\n    CLASS_NAME = np.load('/kaggle/input/bc25-models-pt-files/class_names.npy', allow_pickle=True).tolist(),\n    WINDOW=\"hann\",\n    NFILTER_MEL=128,\n    WINDOW_LENGTH= 1024,\n    WINDOW_STEP= 512,\n    FREQ_HIGH=14000,\n    TARGET_DURATION_S = 5,\n    TARGET_SAMPLES = 5*32000,\n    DEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\"),\n    )\n# Function to seed everything to ensure reproducibility\ndef seed_everything(seed):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False # Change to true if input sizes are kept constant\n\nseed_everything(cfg.SEED)\n# Verifying changes\nprint(cfg.__dict__)\n# Device check\nprint(f\"Using device: {cfg.DEVICE}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T05:03:36.591058Z","iopub.execute_input":"2025-04-30T05:03:36.591417Z","iopub.status.idle":"2025-04-30T05:03:36.618123Z","shell.execute_reply.started":"2025-04-30T05:03:36.591389Z","shell.execute_reply":"2025-04-30T05:03:36.617144Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data pre-processing function","metadata":{}},{"cell_type":"code","source":"\nimport numpy as np\nfrom joblib import Parallel, delayed\nimport tempfile\nfrom tqdm.auto import tqdm\nimport numpy as np\nfrom scipy.signal import filtfilt, fftconvolve, stft, istft\ndef sigmoid(x, shift, mult):\n    \"\"\"\n    Using this sigmoid to discourage one network overpowering the other\n    \"\"\"\n    return 1 / (1 + np.exp(-(x + shift) * mult))\n\ndef _smoothing_filter(n_grad_freq, n_grad_time):\n    \"\"\"Generates a filter to smooth the mask for the spectrogram\n\n    Arguments:\n        n_grad_freq {[type]} -- [how many frequency channels to smooth over with the mask.]\n        n_grad_time {[type]} -- [how many time channels to smooth over with the mask.]\n    \"\"\"\n    smoothing_filter = np.outer(\n        np.concatenate(\n            [\n                np.linspace(0, 1, n_grad_freq + 1, endpoint=False),\n                np.linspace(1, 0, n_grad_freq + 2),\n            ]\n        )[1:-1],\n        np.concatenate(\n            [\n                np.linspace(0, 1, n_grad_time + 1, endpoint=False),\n                np.linspace(1, 0, n_grad_time + 2),\n            ]\n        )[1:-1],\n    )\n    smoothing_filter = smoothing_filter / np.sum(smoothing_filter)\n    return smoothing_filter\n\n\nclass SpectralGate:\n    def __init__(\n            self,\n            y,\n            sr,\n            prop_decrease,\n            chunk_size,\n            padding,\n            n_fft,\n            win_length,\n            hop_length,\n            time_constant_s,\n            freq_mask_smooth_hz,\n            time_mask_smooth_ms,\n            tmp_folder,\n            use_tqdm,\n            n_jobs,\n    ):\n        self.sr = sr\n        # if this is a 1D single channel recording\n        self.flat = False\n\n        y = np.array(y)\n        # reshape data to (#channels, #frames)\n        if len(y.shape) == 1:\n            self.y = np.expand_dims(y, 0)\n            self.flat = True\n        elif len(y.shape) > 2:\n            raise ValueError(\"Waveform must be in shape (# frames, # channels)\")\n        else:\n            self.y = y\n\n        self._dtype = y.dtype\n        # get the number of channels and frames in data\n        self.n_channels, self.n_frames = self.y.shape\n        self._chunk_size = chunk_size\n        self.padding = padding\n        self.n_jobs = n_jobs\n\n        self.use_tqdm = use_tqdm\n        # where to create a temp file for parallel\n        # writing\n        self._tmp_folder = tmp_folder\n\n        ### Parameters for spectral gating\n        self._n_fft = n_fft\n        # set window and hop length for stft\n        if win_length is None:\n            self._win_length = self._n_fft\n        else:\n            self._win_length = win_length\n        if hop_length is None:\n            self._hop_length = self._win_length // 4\n        else:\n            self._hop_length = hop_length\n\n        self._time_constant_s = time_constant_s\n\n        self._prop_decrease = prop_decrease\n\n        if (freq_mask_smooth_hz is None) & (time_mask_smooth_ms is None):\n            self.smooth_mask = False\n        else:\n            self._generate_mask_smoothing_filter(\n                freq_mask_smooth_hz, time_mask_smooth_ms\n            )\n\n    def _generate_mask_smoothing_filter(self, freq_mask_smooth_hz, time_mask_smooth_ms):\n        if freq_mask_smooth_hz is None:\n            n_grad_freq = 1\n        else:\n            # filter to smooth the mask\n            n_grad_freq = int(freq_mask_smooth_hz / (self.sr / (self._n_fft / 2)))\n            if n_grad_freq < 1:\n                raise ValueError(\n                    \"freq_mask_smooth_hz needs to be at least {}Hz\".format(\n                        int((self.sr / (self._n_fft / 2)))\n                    )\n                )\n\n        if time_mask_smooth_ms is None:\n            n_grad_time = 1\n        else:\n            n_grad_time = int(\n                time_mask_smooth_ms / ((self._hop_length / self.sr) * 1000)\n            )\n            if n_grad_time < 1:\n                raise ValueError(\n                    \"time_mask_smooth_ms needs to be at least {}ms\".format(\n                        int((self._hop_length / self.sr) * 1000)\n                    )\n                )\n        if (n_grad_time == 1) & (n_grad_freq == 1):\n            self.smooth_mask = False\n        else:\n            self.smooth_mask = True\n            self._smoothing_filter = _smoothing_filter(n_grad_freq, n_grad_time)\n\n    def _read_chunk(self, i1, i2):\n        \"\"\"read chunk and pad with zerros\"\"\"\n        if i1 < 0:\n            i1b = 0\n        else:\n            i1b = i1\n        if i2 > self.n_frames:\n            i2b = self.n_frames\n        else:\n            i2b = i2\n        chunk = np.zeros((self.n_channels, i2 - i1))\n        chunk[:, i1b - i1: i2b - i1] = self.y[:, i1b:i2b]\n        return chunk\n\n    def filter_chunk(self, start_frame, end_frame):\n        \"\"\"Pad and perform filtering\"\"\"\n        i1 = start_frame - self.padding\n        i2 = end_frame + self.padding\n        padded_chunk = self._read_chunk(i1, i2)\n        filtered_padded_chunk = self._do_filter(padded_chunk)\n        return filtered_padded_chunk[:, start_frame - i1: end_frame - i1]\n\n    def _get_filtered_chunk(self, ind):\n        \"\"\"Grabs a single chunk\"\"\"\n        start0 = ind * self._chunk_size\n        end0 = (ind + 1) * self._chunk_size\n        return self.filter_chunk(start_frame=start0, end_frame=end0)\n\n    def _do_filter(self, chunk):\n        \"\"\"Do the actual filtering\"\"\"\n        raise NotImplementedError\n\n    def _iterate_chunk(self, filtered_chunk, pos, end0, start0, ich):\n        filtered_chunk0 = self._get_filtered_chunk(ich)\n        filtered_chunk[:, pos: pos + end0 - start0] = filtered_chunk0[:, start0:end0]\n        pos += end0 - start0\n\n    def get_traces(self, start_frame=None, end_frame=None):\n        \"\"\"Grab filtered data iterating over chunks\"\"\"\n        if start_frame is None:\n            start_frame = 0\n        if end_frame is None:\n            end_frame = self.n_frames\n\n        if self._chunk_size is not None:\n            if end_frame - start_frame > self._chunk_size:\n                ich1 = int(start_frame / self._chunk_size)\n                ich2 = int((end_frame - 1) / self._chunk_size)\n\n                # write output to temp memmap for parallelization\n                with tempfile.NamedTemporaryFile(prefix=self._tmp_folder) as fp:\n                    # create temp file\n                    filtered_chunk = np.memmap(\n                        fp,\n                        dtype=self._dtype,\n                        shape=(self.n_channels, int(end_frame - start_frame)),\n                        mode=\"w+\",\n                    )\n                    pos_list = []\n                    start_list = []\n                    end_list = []\n                    pos = 0\n                    for ich in range(ich1, ich2 + 1):\n                        if ich == ich1:\n                            start0 = start_frame - ich * self._chunk_size\n                        else:\n                            start0 = 0\n                        if ich == ich2:\n                            end0 = end_frame - ich * self._chunk_size\n                        else:\n                            end0 = self._chunk_size\n                        pos_list.append(pos)\n                        start_list.append(start0)\n                        end_list.append(end0)\n                        pos += end0 - start0\n\n                    Parallel(n_jobs=self.n_jobs)(\n                        delayed(self._iterate_chunk)(\n                            filtered_chunk, pos, end0, start0, ich\n                        )\n                        for pos, start0, end0, ich in zip(\n                            tqdm(pos_list, disable=not (self.use_tqdm)),\n                            start_list,\n                            end_list,\n                            range(ich1, ich2 + 1),\n                        )\n                    )\n                    if self.flat:\n                        return filtered_chunk.astype(self._dtype).flatten()\n                    else:\n                        return filtered_chunk.astype(self._dtype)\n\n        filtered_chunk = self.filter_chunk(start_frame=0, end_frame=end_frame)\n        if self.flat:\n            return filtered_chunk.astype(self._dtype).flatten()\n        else:\n            return filtered_chunk.astype(self._dtype)\n\n\nclass SpectralGateNonStationary(SpectralGate):\n    def __init__(\n            self,\n            y,\n            sr,\n            chunk_size,\n            padding,\n            n_fft,\n            win_length,\n            hop_length,\n            time_constant_s,\n            freq_mask_smooth_hz,\n            time_mask_smooth_ms,\n            thresh_n_mult_nonstationary,\n            sigmoid_slope_nonstationary,\n            tmp_folder,\n            prop_decrease,\n            use_tqdm,\n            n_jobs,\n    ):\n        self._thresh_n_mult_nonstationary = thresh_n_mult_nonstationary\n        self._sigmoid_slope_nonstationary = sigmoid_slope_nonstationary\n\n        super().__init__(\n            y=y,\n            sr=sr,\n            chunk_size=chunk_size,\n            padding=padding,\n            n_fft=n_fft,\n            win_length=win_length,\n            hop_length=hop_length,\n            time_constant_s=time_constant_s,\n            freq_mask_smooth_hz=freq_mask_smooth_hz,\n            time_mask_smooth_ms=time_mask_smooth_ms,\n            tmp_folder=tmp_folder,\n            prop_decrease=prop_decrease,\n            use_tqdm=use_tqdm,\n            n_jobs=n_jobs,\n        )\n\n    def spectral_gating_nonstationary(self, chunk):\n        \"\"\"non-stationary version of spectral gating\"\"\"\n        denoised_channels = np.zeros(chunk.shape, chunk.dtype)\n        for ci, channel in enumerate(chunk):\n            _, _, sig_stft = stft(\n                channel,\n                nfft=self._n_fft,\n                noverlap=self._win_length - self._hop_length,\n                nperseg=self._win_length,\n                padded=False\n            )\n            # get abs of signal stft\n            abs_sig_stft = np.abs(sig_stft)\n\n            # get the smoothed mean of the signal\n            sig_stft_smooth = get_time_smoothed_representation(\n                abs_sig_stft,\n                self.sr,\n                self._hop_length,\n                time_constant_s=self._time_constant_s,\n            )\n\n            # get the number of X above the mean the signal is\n            sig_mult_above_thresh = (abs_sig_stft - sig_stft_smooth) / sig_stft_smooth\n            # mask based on sigmoid\n            sig_mask = sigmoid(\n                sig_mult_above_thresh,\n                -self._thresh_n_mult_nonstationary,\n                self._sigmoid_slope_nonstationary,\n            )\n\n            if self.smooth_mask:\n                # convolve the mask with a smoothing filter\n                sig_mask = fftconvolve(sig_mask, self._smoothing_filter, mode=\"same\")\n\n            sig_mask = sig_mask * self._prop_decrease + np.ones(np.shape(sig_mask)) * (\n                    1.0 - self._prop_decrease\n            )\n\n            # multiply signal with mask\n            sig_stft_denoised = sig_stft * sig_mask\n\n            # invert/recover the signal\n            _, denoised_signal = istft(\n                sig_stft_denoised,\n                nfft=self._n_fft,\n                noverlap=self._win_length - self._hop_length,\n                nperseg=self._win_length\n            )\n            denoised_channels[ci, : len(denoised_signal)] = denoised_signal\n        return denoised_channels\n\n    def _do_filter(self, chunk):\n        \"\"\"Do the actual filtering\"\"\"\n        chunk_filtered = self.spectral_gating_nonstationary(chunk)\n\n        return chunk_filtered\n\n\ndef get_time_smoothed_representation(\n        spectral, samplerate, hop_length, time_constant_s=0.001\n):\n    t_frames = time_constant_s * samplerate / float(hop_length)\n    # By default, this solves the equation for b:\n    #   b**2  + (1 - b) / t_frames  - 2 = 0\n    # which approximates the full-width half-max of the\n    # squared frequency response of the IIR low-pass filt\n    b = (np.sqrt(1 + 4 * t_frames ** 2) - 1) / (2 * t_frames ** 2)\n    return filtfilt([b], [1, b - 1], spectral, axis=-1, padtype=None)\n\n\ndef reduce_noise(\n        y,\n        sr,\n        stationary=False,\n        y_noise=None,\n        prop_decrease=1.0,\n        time_constant_s=2.0,\n        freq_mask_smooth_hz=500,\n        time_mask_smooth_ms=50,\n        thresh_n_mult_nonstationary=2,\n        sigmoid_slope_nonstationary=10,\n        n_std_thresh_stationary=1.5,\n        tmp_folder=None,\n        chunk_size=600000,\n        padding=30000,\n        n_fft=1024,\n        win_length=None,\n        hop_length=None,\n        clip_noise_stationary=True,\n        use_tqdm=False,\n        n_jobs=1,\n        use_torch=False,\n        device=\"cuda\",\n):\n    sg = SpectralGateNonStationary(\n        y=y,\n        sr=sr,\n        chunk_size=chunk_size,\n        padding=padding,\n        prop_decrease=prop_decrease,\n        n_fft=n_fft,\n        win_length=win_length,\n        hop_length=hop_length,\n        time_constant_s=time_constant_s,\n        freq_mask_smooth_hz=freq_mask_smooth_hz,\n        time_mask_smooth_ms=time_mask_smooth_ms,\n        thresh_n_mult_nonstationary=thresh_n_mult_nonstationary,\n        sigmoid_slope_nonstationary=sigmoid_slope_nonstationary,\n        tmp_folder=tmp_folder,\n        use_tqdm=use_tqdm,\n        n_jobs=n_jobs,\n    )\n    return sg.get_traces()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T05:03:43.091380Z","iopub.execute_input":"2025-04-30T05:03:43.091716Z","iopub.status.idle":"2025-04-30T05:03:43.127036Z","shell.execute_reply.started":"2025-04-30T05:03:43.091693Z","shell.execute_reply":"2025-04-30T05:03:43.126098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================\n# 1. Hàm load và tiền xử lý audio\n# ============================\ndef load_and_preprocess(audio_path, sr):\n    \"\"\"\n    - Load file .ogg bằng librosa.\n    - Giảm nhiễu bằng noisereduce.\n    - Tăng âm lượng bằng pedalboard.\n    \"\"\"\n    #he so khuech dai\n    gain_db = 20\n    gain_linear = 10 ** (gain_db / 20)\n    samples, sr = librosa.load(audio_path, sr=sr)\n    samples_nr = reduce_noise(y=samples, sr=sr)\n    samples_nr = samples_nr * gain_linear\n    # board = Pedalboard([Gain(gain_db=10)])\n    # samples_proc = board(samples_nr, sr)\n    return samples_nr, sr\n\n# ============================\n# 2. Hàm tạo sliding window (5 giây, bước nhảy 0.5 giây)\n# ============================\ndef get_sliding_windows(audio, sr, segment_duration=5.0, overlap_percent=0.5):\n    \"\"\"\n    Trả về danh sách các tuple (start_sample, end_sample) cho mỗi window.\n    \"\"\"\n    step = segment_duration*(1-overlap_percent)\n    seg_samples = int(segment_duration * sr)\n    step_samples = int(step * sr)\n    windows = []\n    for start in range(0, len(audio) - seg_samples + 1, step_samples):\n        end = start + seg_samples\n        windows.append((start, end))\n    return windows\n\n# -------------------------\n# 4. Tạo Mel Spectrogram\n# -------------------------\n\ndef compute_mel_spectrogram_librosa(\n    audio_segment: np.ndarray,\n    cfg\n) -> np.ndarray:\n    # 1) Compute power spectrogram with STFT\n    S = librosa.stft(\n        audio_segment,\n        n_fft=cfg.WINDOW_LENGTH,\n        hop_length=cfg.WINDOW_STEP,\n        win_length=cfg.WINDOW_LENGTH,\n        window='hann',\n        center=True,\n        pad_mode='reflect'\n    )\n    # Convert to power (magnitude squared)\n    S_power = np.abs(S)**2\n\n    # 2) Apply Mel filter bank\n    mel_S = librosa.feature.melspectrogram(\n        S=S_power,\n        sr=cfg.SAMPLE_RATE,\n        n_mels=cfg.NFILTER_MEL,\n        fmax=cfg.FREQ_HIGH\n    )\n\n    # 3) Convert to decibel scale\n    S_db = librosa.power_to_db(mel_S, ref=np.max, top_db=80.0)\n\n    return S_db\n    \n# ============================\n# 5. Lưu ảnh Mel Spectrogram\n# ============================\nfrom PIL import Image\nimport matplotlib.cm as cm\n\ndef get_spectrogram_image(mel_spec, cfg, output_size=(1280, 720)):\n    mel_spec = (mel_spec - mel_spec.min()) / (mel_spec.max() - mel_spec.min())\n    # List of colormaps\n    cmap_name = cfg.COLOR_MAP\n    cmap = cm.get_cmap(cmap_name)\n    rgba_img = cmap(mel_spec)  # shape: (H, W, 4), values in [0, 1]\n    rgb_img = (rgba_img[:, :, :3] * 255).astype(np.uint8)\n    img = Image.fromarray(rgb_img)\n    img = img.transpose(Image.FLIP_TOP_BOTTOM)\n    img_resized = img.resize(output_size, Image.LANCZOS)\n    return img_resized\n    \ndef show_spectrogram(spec, title, sr, hop_length, y_axis=\"log\", x_axis=\"time\"):\n    librosa.display.specshow(\n        spec, sr=sr, y_axis=y_axis, x_axis=x_axis, hop_length=hop_length\n    )\n    plt.title(title)\n    plt.colorbar(format=\"%+2.0f dB\")\n    plt.tight_layout()\n    plt.show()\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T05:09:40.555126Z","iopub.execute_input":"2025-04-30T05:09:40.556179Z","iopub.status.idle":"2025-04-30T05:09:40.568551Z","shell.execute_reply.started":"2025-04-30T05:09:40.556139Z","shell.execute_reply":"2025-04-30T05:09:40.567175Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from typing import Dict, Optional\nfrom transformers import (\n    AutoConfig,\n    ConvNextConfig,\n    ConvNextForImageClassification,\n    ConvNextModel,\n)\n\n\n\nclass ConvNextClassifier(nn.Module):\n    def __init__(\n        self,\n        num_channels: int = 1,\n        embedding_size: Optional[int] = None,\n    ):\n        super().__init__()\n        self.num_channels = num_channels\n        self.embedding_size = embedding_size\n        self.model = None\n        self.architecture = None\n        self._initialize_model()\n\n    def _initialize_model(self) -> nn.Module:\n        \"\"\"Initializes the ConvNext model based on specified attributes.\n\n        Returns:\n            nn.Module: The initialized ConvNext model.\n        \"\"\"\n\n        adjusted_state_dict = None\n\n        model = ConvNextModel\n        config = ConvNextConfig.from_pretrained(\"/kaggle/input/bc25-models-pt-files/convnext_base_facebook_224_22k\")\n        hidden_sizes = config.hidden_sizes\n        hidden_sizes[-1] = self.embedding_size\n        self.model = model.from_pretrained(\n                \"/kaggle/input/bc25-models-pt-files/convnext_base_facebook_224_22k\",\n                hidden_sizes=hidden_sizes,\n                num_channels=self.num_channels,\n                cache_dir=None,\n                ignore_mismatched_sizes=True,\n            )\n\n        self.architecture = self.model.config.model_type\n\n    def forward(self, input_values: torch.Tensor, labels: Optional[torch.Tensor] = None\n    ) -> torch.Tensor:\n\n        outputs = self.model(input_values)\n\n        output = outputs.last_hidden_state\n\n        return output\n        \nclass CustomNet(nn.Module):\n    def __init__(self, backbone, num_classes, add_on_layers_type=\"identity\", use_prototype = False, num_prototypes = 5):\n        super().__init__()\n        self.backbone = backbone\n        backbone_out = backbone.embedding_size\n        if add_on_layers_type == \"identity\":\n            self.add_on_layers = nn.Sequential(nn.Identity())\n        elif add_on_layers_type == \"upsample\":\n            self.add_on_layers = nn.Upsample(scale_factor=2, mode=\"bilinear\")\n        self.pooling = nn.AdaptiveAvgPool2d(1)\n        self.classifier = nn.Sequential(\n            self.add_on_layers,\n            nn.Linear(backbone_out, num_classes)\n        )\n        \n    def forward(self,x):\n        features = self.backbone(x)\n        if features.dim() == 4:\n            features = self.pooling(features)\n            features = features.view(features.size(0), -1)\n        \n        logits = self.classifier(features)\n        return logits","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T05:04:10.737482Z","iopub.execute_input":"2025-04-30T05:04:10.737816Z","iopub.status.idle":"2025-04-30T05:04:10.750029Z","shell.execute_reply.started":"2025-04-30T05:04:10.737791Z","shell.execute_reply":"2025-04-30T05:04:10.748681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision import transforms\ndata_transforms = {\n    'test': transforms.Compose([\n        transforms.Resize(224),\n        transforms.Grayscale(num_output_channels=1),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.5], std=[0.5])\n    ])\n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T05:04:16.216264Z","iopub.execute_input":"2025-04-30T05:04:16.216628Z","iopub.status.idle":"2025-04-30T05:04:16.222511Z","shell.execute_reply.started":"2025-04-30T05:04:16.216601Z","shell.execute_reply":"2025-04-30T05:04:16.221329Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"backbone = ConvNextClassifier(\n    num_channels=1,\n    embedding_size=1024,\n)\nmodel = CustomNet(backbone, cfg.NUM_CLASSES).to(cfg.DEVICE)\ncheckpoint = torch.load(\"/kaggle/input/bc25-models-pt-files/latest_model.pth\", map_location=cfg.DEVICE)\n\nstate_dict_full= checkpoint['model_state_dict']\nprefix = '_orig_mod.'\nmodel_dict = {}\nfor key, val in state_dict_full.items():\n    new_key = key[len(prefix):]\n    model_dict[new_key] = val\nmodel.load_state_dict(model_dict, strict=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T05:04:19.831922Z","iopub.execute_input":"2025-04-30T05:04:19.832264Z","iopub.status.idle":"2025-04-30T05:04:31.957012Z","shell.execute_reply.started":"2025-04-30T05:04:19.832240Z","shell.execute_reply":"2025-04-30T05:04:31.956038Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# not only visualize, but also pre-allocated\nsig, rate = load_and_preprocess(\"/kaggle/input/birdclef-2025/train_soundscapes/H02_20230420_074000.ogg\", 32000)\nchunk = sig[32000*0:32000*5]\nsegment = np.array(chunk, dtype=np.float32)\nmel_spec_db = compute_mel_spectrogram_librosa(segment, cfg)\npil_img = get_spectrogram_image(mel_spec_db,cfg,(640,480))\ndisplay(pil_img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-30T05:09:49.962691Z","iopub.execute_input":"2025-04-30T05:09:49.963059Z","iopub.status.idle":"2025-04-30T05:09:51.632836Z","shell.execute_reply.started":"2025-04-30T05:09:49.963034Z","shell.execute_reply":"2025-04-30T05:09:51.631869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import time\n\n# Set seed\nnp.random.seed(42)\n\n# sample submission:\nsubmission_example = pd.read_csv(cfg.SUBMISSION_SAMPLE_PATH)\n# Class labels from train audio\nclass_labels = sorted(cfg.CLASS_NAME)\n\n# List of test soundscapes (only visible during submission)\ntest_soundscape_path = '/kaggle/input/birdclef-2025/test_soundscapes/'\ntest_soundscapes = [os.path.join(test_soundscape_path, afile) for afile in sorted(os.listdir(test_soundscape_path)) if afile.endswith('.ogg')]\n\n# Open each soundscape and make predictions for 5-second segments\n# Use pandas df with 'row_id' plus class labels as columns\n# start = time.time()\n# Prepare empty list for rows\nrows = []\nfor soundscape in test_soundscapes:\n\n    # Load audio\n    sig, rate = load_and_preprocess(soundscape, 32000)\n\n        \n    # Split into 5-second chunks\n    for i in range(0, len(sig), rate * 5):\n        chunk = sig[i:i + rate * 5]\n        if len(chunk) == 0:\n            continue  # Skip empty chunks\n        # Get row id  (soundscape id + end time of 5s chunk)      \n        row_id = os.path.basename(soundscape).split('.')[0] + f'_{i//rate + 5}'\n        \n        # Preprocessing (\n        segment = np.array(chunk, dtype=np.float32)\n        mel_spec_db = compute_mel_spectrogram_librosa(segment, cfg)\n        pil_img = get_spectrogram_image(mel_spec_db,cfg,(640,480))\n        x = data_transforms['test'](pil_img).unsqueeze(0) # (1, C, H, W)\n        # Model prediction\n        with torch.no_grad():\n            logits = model(x)  \n            scores = torch.sigmoid(logits)  \n            scores = scores.squeeze(0).cpu().numpy()  \n        \n        # Append to predictions as new row\n        row_data = [row_id] + list(scores)\n        rows.append(row_data)\n\n# end = time.time()\n# Build predictions DataFrame\npredictions = pd.DataFrame(rows, columns=['row_id'] + class_labels)\n\n# Now remap to submission columns\nsubmission_cols = submission_example.columns[1:]\n\n# Reindex predictions to match submission\npredictions = predictions[['row_id'] + list(submission_cols)]\npredictions.to_csv('submission.csv', index=False)\npredictions.head()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-30T05:09:58.151151Z","iopub.execute_input":"2025-04-30T05:09:58.151441Z","iopub.status.idle":"2025-04-30T05:10:24.056188Z","shell.execute_reply.started":"2025-04-30T05:09:58.151417Z","shell.execute_reply":"2025-04-30T05:10:24.054875Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# mean_time_per_file = (end - start)/10\n# estimate_total_time = mean_time_per_file*800\n# print(\"mean time per file: \",mean_time_per_file)\n# print(\"estimate total time: \",estimate_total_time)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-29T03:47:06.248886Z","iopub.execute_input":"2025-04-29T03:47:06.249224Z","iopub.status.idle":"2025-04-29T03:47:06.254629Z","shell.execute_reply.started":"2025-04-29T03:47:06.249197Z","shell.execute_reply":"2025-04-29T03:47:06.253616Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}