{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11606750,"sourceType":"datasetVersion","datasetId":7277400},{"sourceId":11633037,"sourceType":"datasetVersion","datasetId":7298741},{"sourceId":11656859,"sourceType":"datasetVersion","datasetId":7315261},{"sourceId":236154035,"sourceType":"kernelVersion"}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Import and config","metadata":{}},{"cell_type":"code","source":"!cp /kaggle/input/nvidia-dali-installation-package/nvidia-dali/* .\n!pip install --no-index --find-links=. \\\n    nvidia_nvjpeg_cu12*.whl \\\n    nvidia_nvjpeg2k_cu12*.whl \\\n    nvidia_nvtiff_cu12*.whl \\\n    nvidia_nvimgcodec_cu12*.whl \n!cd /kaggle/input/nvidia-dali-installation-package/nvidia-dali && ls |grep nvidia_dali_nightly_cuda120 |xargs pip install \nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport os, gc, random \nimport pandas as pd\nimport pickle\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\nimport IPython.display as ipd\nfrom IPython.display import display, clear_output\nimport ipywidgets as widgets\nimport librosa\nimport librosa.display\nimport soundfile as sf\nimport numpy as np\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, accuracy_score, confusion_matrix\n\nfrom nvidia.dali import pipeline_def\nimport nvidia.dali.fn as fn\nimport nvidia.dali.types as types\nimport nvidia.dali as dali\n\nclear_output()\nprint(\"Install and Import DONE\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:07:14.804883Z","iopub.execute_input":"2025-05-03T05:07:14.805518Z","iopub.status.idle":"2025-05-03T05:07:23.057514Z","shell.execute_reply.started":"2025-05-03T05:07:14.805492Z","shell.execute_reply":"2025-05-03T05:07:23.056629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nclass Config:\n    def __init__(self, **kwargs):\n        for k, v in kwargs.items():\n            setattr(self, k, v)\n\n    def update(self, **kwargs):\n        for k, v in kwargs.items():\n            setattr(self, k, v)\n\n# Initialize and set basic configuration\ncfg = Config(\n    SEED=42, \n    USE_AUDIO_AS_INPUT = False, # tiền xử lý lại từ đầu nếu set True, set False sẽ lấy đầu vào là ảnh.\n    USE_SMOOTH_LABEL = True,\n    MEATA_DATA_PATH = Path('/kaggle/input/sounscape-bc25/meta_train_soundscape.csv'),\n    SUBMISSION_SAMPLE_PATH = Path('/kaggle/input/birdclef-2025/sample_submission.csv'),\n    DATA_PATH=Path(\"/kaggle/input/sounscape-bc25/train_soundscapes_data\"),\n    OUTPUT_FOLDER =Path(\"/kaggle/working/\"),\n    MODEL_PATH = Path(\"/kaggle/input/bc25-models-pt-files/latest_model.pth\"),\n    CLASS_NAME = np.load('/kaggle/input/bc25-models-pt-files/class_names.npy', allow_pickle=True).tolist(),\n    NUM_CLASSES = 206,\n    COLOR_MAP ='inferno',\n    SAMPLE_RATE=32000,\n    WINDOW=\"hann\",\n    NFILTER_MEL=128,\n    WINDOW_LENGTH= 1024,\n    WINDOW_STEP= 512,\n    FREQ_HIGH=14000,\n    TARGET_DURATION_S = 5,\n    TARGET_SAMPLES = 5*32000,\n    DEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\"),\n    )\n# Function to seed everything to ensure reproducibility\ndef seed_everything(seed):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False # Change to true if input sizes are kept constant\n\nseed_everything(cfg.SEED)\n# Verifying changes\nprint(cfg.__dict__)\n# Device check\nprint(f\"Using device: {cfg.DEVICE}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:22:20.451086Z","iopub.execute_input":"2025-05-03T05:22:20.451739Z","iopub.status.idle":"2025-05-03T05:22:20.466262Z","shell.execute_reply.started":"2025-05-03T05:22:20.451714Z","shell.execute_reply":"2025-05-03T05:22:20.465482Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data pre-processing function","metadata":{}},{"cell_type":"code","source":"# ==============This cell is a copy of noice reduce lib ====================\nimport numpy as np\nfrom joblib import Parallel, delayed\nimport tempfile\nfrom tqdm.auto import tqdm\nimport numpy as np\nfrom scipy.signal import filtfilt, fftconvolve, stft, istft\ndef sigmoid(x, shift, mult):\n    \"\"\"\n    Using this sigmoid to discourage one network overpowering the other\n    \"\"\"\n    return 1 / (1 + np.exp(-(x + shift) * mult))\n\ndef _smoothing_filter(n_grad_freq, n_grad_time):\n    \"\"\"Generates a filter to smooth the mask for the spectrogram\n\n    Arguments:\n        n_grad_freq {[type]} -- [how many frequency channels to smooth over with the mask.]\n        n_grad_time {[type]} -- [how many time channels to smooth over with the mask.]\n    \"\"\"\n    smoothing_filter = np.outer(\n        np.concatenate(\n            [\n                np.linspace(0, 1, n_grad_freq + 1, endpoint=False),\n                np.linspace(1, 0, n_grad_freq + 2),\n            ]\n        )[1:-1],\n        np.concatenate(\n            [\n                np.linspace(0, 1, n_grad_time + 1, endpoint=False),\n                np.linspace(1, 0, n_grad_time + 2),\n            ]\n        )[1:-1],\n    )\n    smoothing_filter = smoothing_filter / np.sum(smoothing_filter)\n    return smoothing_filter\n\n\nclass SpectralGate:\n    def __init__(\n            self,\n            y,\n            sr,\n            prop_decrease,\n            chunk_size,\n            padding,\n            n_fft,\n            win_length,\n            hop_length,\n            time_constant_s,\n            freq_mask_smooth_hz,\n            time_mask_smooth_ms,\n            tmp_folder,\n            use_tqdm,\n            n_jobs,\n    ):\n        self.sr = sr\n        # if this is a 1D single channel recording\n        self.flat = False\n\n        y = np.array(y)\n        # reshape data to (#channels, #frames)\n        if len(y.shape) == 1:\n            self.y = np.expand_dims(y, 0)\n            self.flat = True\n        elif len(y.shape) > 2:\n            raise ValueError(\"Waveform must be in shape (# frames, # channels)\")\n        else:\n            self.y = y\n\n        self._dtype = y.dtype\n        # get the number of channels and frames in data\n        self.n_channels, self.n_frames = self.y.shape\n        self._chunk_size = chunk_size\n        self.padding = padding\n        self.n_jobs = n_jobs\n\n        self.use_tqdm = use_tqdm\n        # where to create a temp file for parallel\n        # writing\n        self._tmp_folder = tmp_folder\n\n        ### Parameters for spectral gating\n        self._n_fft = n_fft\n        # set window and hop length for stft\n        if win_length is None:\n            self._win_length = self._n_fft\n        else:\n            self._win_length = win_length\n        if hop_length is None:\n            self._hop_length = self._win_length // 4\n        else:\n            self._hop_length = hop_length\n\n        self._time_constant_s = time_constant_s\n\n        self._prop_decrease = prop_decrease\n\n        if (freq_mask_smooth_hz is None) & (time_mask_smooth_ms is None):\n            self.smooth_mask = False\n        else:\n            self._generate_mask_smoothing_filter(\n                freq_mask_smooth_hz, time_mask_smooth_ms\n            )\n\n    def _generate_mask_smoothing_filter(self, freq_mask_smooth_hz, time_mask_smooth_ms):\n        if freq_mask_smooth_hz is None:\n            n_grad_freq = 1\n        else:\n            # filter to smooth the mask\n            n_grad_freq = int(freq_mask_smooth_hz / (self.sr / (self._n_fft / 2)))\n            if n_grad_freq < 1:\n                raise ValueError(\n                    \"freq_mask_smooth_hz needs to be at least {}Hz\".format(\n                        int((self.sr / (self._n_fft / 2)))\n                    )\n                )\n\n        if time_mask_smooth_ms is None:\n            n_grad_time = 1\n        else:\n            n_grad_time = int(\n                time_mask_smooth_ms / ((self._hop_length / self.sr) * 1000)\n            )\n            if n_grad_time < 1:\n                raise ValueError(\n                    \"time_mask_smooth_ms needs to be at least {}ms\".format(\n                        int((self._hop_length / self.sr) * 1000)\n                    )\n                )\n        if (n_grad_time == 1) & (n_grad_freq == 1):\n            self.smooth_mask = False\n        else:\n            self.smooth_mask = True\n            self._smoothing_filter = _smoothing_filter(n_grad_freq, n_grad_time)\n\n    def _read_chunk(self, i1, i2):\n        \"\"\"read chunk and pad with zerros\"\"\"\n        if i1 < 0:\n            i1b = 0\n        else:\n            i1b = i1\n        if i2 > self.n_frames:\n            i2b = self.n_frames\n        else:\n            i2b = i2\n        chunk = np.zeros((self.n_channels, i2 - i1))\n        chunk[:, i1b - i1: i2b - i1] = self.y[:, i1b:i2b]\n        return chunk\n\n    def filter_chunk(self, start_frame, end_frame):\n        \"\"\"Pad and perform filtering\"\"\"\n        i1 = start_frame - self.padding\n        i2 = end_frame + self.padding\n        padded_chunk = self._read_chunk(i1, i2)\n        filtered_padded_chunk = self._do_filter(padded_chunk)\n        return filtered_padded_chunk[:, start_frame - i1: end_frame - i1]\n\n    def _get_filtered_chunk(self, ind):\n        \"\"\"Grabs a single chunk\"\"\"\n        start0 = ind * self._chunk_size\n        end0 = (ind + 1) * self._chunk_size\n        return self.filter_chunk(start_frame=start0, end_frame=end0)\n\n    def _do_filter(self, chunk):\n        \"\"\"Do the actual filtering\"\"\"\n        raise NotImplementedError\n\n    def _iterate_chunk(self, filtered_chunk, pos, end0, start0, ich):\n        filtered_chunk0 = self._get_filtered_chunk(ich)\n        filtered_chunk[:, pos: pos + end0 - start0] = filtered_chunk0[:, start0:end0]\n        pos += end0 - start0\n\n    def get_traces(self, start_frame=None, end_frame=None):\n        \"\"\"Grab filtered data iterating over chunks\"\"\"\n        if start_frame is None:\n            start_frame = 0\n        if end_frame is None:\n            end_frame = self.n_frames\n\n        if self._chunk_size is not None:\n            if end_frame - start_frame > self._chunk_size:\n                ich1 = int(start_frame / self._chunk_size)\n                ich2 = int((end_frame - 1) / self._chunk_size)\n\n                # write output to temp memmap for parallelization\n                with tempfile.NamedTemporaryFile(prefix=self._tmp_folder) as fp:\n                    # create temp file\n                    filtered_chunk = np.memmap(\n                        fp,\n                        dtype=self._dtype,\n                        shape=(self.n_channels, int(end_frame - start_frame)),\n                        mode=\"w+\",\n                    )\n                    pos_list = []\n                    start_list = []\n                    end_list = []\n                    pos = 0\n                    for ich in range(ich1, ich2 + 1):\n                        if ich == ich1:\n                            start0 = start_frame - ich * self._chunk_size\n                        else:\n                            start0 = 0\n                        if ich == ich2:\n                            end0 = end_frame - ich * self._chunk_size\n                        else:\n                            end0 = self._chunk_size\n                        pos_list.append(pos)\n                        start_list.append(start0)\n                        end_list.append(end0)\n                        pos += end0 - start0\n\n                    Parallel(n_jobs=self.n_jobs)(\n                        delayed(self._iterate_chunk)(\n                            filtered_chunk, pos, end0, start0, ich\n                        )\n                        for pos, start0, end0, ich in zip(\n                            tqdm(pos_list, disable=not (self.use_tqdm)),\n                            start_list,\n                            end_list,\n                            range(ich1, ich2 + 1),\n                        )\n                    )\n                    if self.flat:\n                        return filtered_chunk.astype(self._dtype).flatten()\n                    else:\n                        return filtered_chunk.astype(self._dtype)\n\n        filtered_chunk = self.filter_chunk(start_frame=0, end_frame=end_frame)\n        if self.flat:\n            return filtered_chunk.astype(self._dtype).flatten()\n        else:\n            return filtered_chunk.astype(self._dtype)\n\n\nclass SpectralGateNonStationary(SpectralGate):\n    def __init__(\n            self,\n            y,\n            sr,\n            chunk_size,\n            padding,\n            n_fft,\n            win_length,\n            hop_length,\n            time_constant_s,\n            freq_mask_smooth_hz,\n            time_mask_smooth_ms,\n            thresh_n_mult_nonstationary,\n            sigmoid_slope_nonstationary,\n            tmp_folder,\n            prop_decrease,\n            use_tqdm,\n            n_jobs,\n    ):\n        self._thresh_n_mult_nonstationary = thresh_n_mult_nonstationary\n        self._sigmoid_slope_nonstationary = sigmoid_slope_nonstationary\n\n        super().__init__(\n            y=y,\n            sr=sr,\n            chunk_size=chunk_size,\n            padding=padding,\n            n_fft=n_fft,\n            win_length=win_length,\n            hop_length=hop_length,\n            time_constant_s=time_constant_s,\n            freq_mask_smooth_hz=freq_mask_smooth_hz,\n            time_mask_smooth_ms=time_mask_smooth_ms,\n            tmp_folder=tmp_folder,\n            prop_decrease=prop_decrease,\n            use_tqdm=use_tqdm,\n            n_jobs=n_jobs,\n        )\n\n    def spectral_gating_nonstationary(self, chunk):\n        \"\"\"non-stationary version of spectral gating\"\"\"\n        denoised_channels = np.zeros(chunk.shape, chunk.dtype)\n        for ci, channel in enumerate(chunk):\n            _, _, sig_stft = stft(\n                channel,\n                nfft=self._n_fft,\n                noverlap=self._win_length - self._hop_length,\n                nperseg=self._win_length,\n                padded=False\n            )\n            # get abs of signal stft\n            abs_sig_stft = np.abs(sig_stft)\n\n            # get the smoothed mean of the signal\n            sig_stft_smooth = get_time_smoothed_representation(\n                abs_sig_stft,\n                self.sr,\n                self._hop_length,\n                time_constant_s=self._time_constant_s,\n            )\n\n            # get the number of X above the mean the signal is\n            sig_mult_above_thresh = (abs_sig_stft - sig_stft_smooth) / sig_stft_smooth\n            # mask based on sigmoid\n            sig_mask = sigmoid(\n                sig_mult_above_thresh,\n                -self._thresh_n_mult_nonstationary,\n                self._sigmoid_slope_nonstationary,\n            )\n\n            if self.smooth_mask:\n                # convolve the mask with a smoothing filter\n                sig_mask = fftconvolve(sig_mask, self._smoothing_filter, mode=\"same\")\n\n            sig_mask = sig_mask * self._prop_decrease + np.ones(np.shape(sig_mask)) * (\n                    1.0 - self._prop_decrease\n            )\n\n            # multiply signal with mask\n            sig_stft_denoised = sig_stft * sig_mask\n\n            # invert/recover the signal\n            _, denoised_signal = istft(\n                sig_stft_denoised,\n                nfft=self._n_fft,\n                noverlap=self._win_length - self._hop_length,\n                nperseg=self._win_length\n            )\n            denoised_channels[ci, : len(denoised_signal)] = denoised_signal\n        return denoised_channels\n\n    def _do_filter(self, chunk):\n        \"\"\"Do the actual filtering\"\"\"\n        chunk_filtered = self.spectral_gating_nonstationary(chunk)\n\n        return chunk_filtered\n\n\ndef get_time_smoothed_representation(\n        spectral, samplerate, hop_length, time_constant_s=0.001\n):\n    t_frames = time_constant_s * samplerate / float(hop_length)\n    # By default, this solves the equation for b:\n    #   b**2  + (1 - b) / t_frames  - 2 = 0\n    # which approximates the full-width half-max of the\n    # squared frequency response of the IIR low-pass filt\n    b = (np.sqrt(1 + 4 * t_frames ** 2) - 1) / (2 * t_frames ** 2)\n    return filtfilt([b], [1, b - 1], spectral, axis=-1, padtype=None)\n\n\ndef reduce_noise(\n        y,\n        sr,\n        stationary=False,\n        y_noise=None,\n        prop_decrease=1.0,\n        time_constant_s=2.0,\n        freq_mask_smooth_hz=500,\n        time_mask_smooth_ms=50,\n        thresh_n_mult_nonstationary=2,\n        sigmoid_slope_nonstationary=10,\n        n_std_thresh_stationary=1.5,\n        tmp_folder=None,\n        chunk_size=600000,\n        padding=30000,\n        n_fft=1024,\n        win_length=None,\n        hop_length=None,\n        clip_noise_stationary=True,\n        use_tqdm=False,\n        n_jobs=1,\n        use_torch=False,\n        device=\"cuda\",\n):\n    sg = SpectralGateNonStationary(\n        y=y,\n        sr=sr,\n        chunk_size=chunk_size,\n        padding=padding,\n        prop_decrease=prop_decrease,\n        n_fft=n_fft,\n        win_length=win_length,\n        hop_length=hop_length,\n        time_constant_s=time_constant_s,\n        freq_mask_smooth_hz=freq_mask_smooth_hz,\n        time_mask_smooth_ms=time_mask_smooth_ms,\n        thresh_n_mult_nonstationary=thresh_n_mult_nonstationary,\n        sigmoid_slope_nonstationary=sigmoid_slope_nonstationary,\n        tmp_folder=tmp_folder,\n        use_tqdm=use_tqdm,\n        n_jobs=n_jobs,\n    )\n    return sg.get_traces()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:07:23.074582Z","iopub.execute_input":"2025-05-03T05:07:23.074848Z","iopub.status.idle":"2025-05-03T05:07:23.251434Z","shell.execute_reply.started":"2025-05-03T05:07:23.074830Z","shell.execute_reply":"2025-05-03T05:07:23.250532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision import transforms\n# ============= Data transform to convert 2 tensor==============\ndata_transforms = {\n    'test': transforms.Compose([\n        transforms.Resize(224),\n        transforms.Grayscale(num_output_channels=1),\n        transforms.ToTensor(),\n        transforms.Normalize(mean=[0.5], std=[0.5])\n    ])\n}\n# ============================\n# 1. Hàm load và tiền xử lý audio\n# ============================\ndef load_and_preprocess(audio_path, sr):\n    \"\"\"\n    - Load file .ogg bằng librosa.\n    - Giảm nhiễu bằng noisereduce.\n    - Tăng âm lượng bằng pedalboard.\n    \"\"\"\n    samples, sr = librosa.load(audio_path, sr=sr)\n    samples_nr = reduce_noise(y=samples, sr=sr)\n    #apply 10dB gain\n    gain_db = 10\n    gain_linear = 10**(gain_db / 20)\n    samples_proc = samples_nr*gain_linear\n    return samples_nr, sr\n\n# ============================\n# 2. Hàm tạo sliding window (5 giây, bước nhảy 0.5 giây)\n# ============================\ndef get_sliding_windows(audio, sr, segment_duration=5.0, overlap_percent=0.5):\n    \"\"\"\n    Trả về danh sách các tuple (start_sample, end_sample) cho mỗi window.\n    \"\"\"\n    step = segment_duration*(1-overlap_percent)\n    seg_samples = int(segment_duration * sr)\n    step_samples = int(step * sr)\n    windows = []\n    for start in range(0, len(audio) - seg_samples + 1, step_samples):\n        end = start + seg_samples\n        windows.append((start, end))\n    return windows\n\n# -------------------------\n# 3. Tạo Mel Spectrogram\n# -------------------------\n\ndef compute_mel_spectrogram(audio_segment, cfg):\n    audio_data = np.array(audio_segment, dtype=np.float32)\n\n    @pipeline_def\n    def mel_spectrogram_pipe(nfft, window_length, window_step, sample_rate, nfilter, freq_high, device=\"cpu\"):\n        audio = types.Constant(device=device, value=audio_data)\n        spectrogram = fn.spectrogram(\n            audio,\n            device=device,\n            nfft=nfft,\n            window_length=window_length,\n            window_step=window_step,\n        )\n        mel_spectrogram = fn.mel_filter_bank(\n            spectrogram,\n            device=device,\n            sample_rate=sample_rate,\n            nfilter=nfilter,\n            freq_high=freq_high\n        )\n        mel_spectrogram_dB = fn.to_decibels(\n            mel_spectrogram,\n            device=device,\n            multiplier=10.0,\n            cutoff_db=-80\n        )\n        return mel_spectrogram_dB\n\n    pipe = mel_spectrogram_pipe(\n        device=\"cpu\",\n        batch_size=1,\n        num_threads=1,\n        nfft=cfg.WINDOW_LENGTH,\n        window_length=cfg.WINDOW_LENGTH,\n        window_step=cfg.WINDOW_STEP,\n        sample_rate=cfg.SAMPLE_RATE,\n        nfilter=cfg.NFILTER_MEL,\n        freq_high=cfg.FREQ_HIGH,\n    )\n\n    pipe.build()\n    outputs = pipe.run()\n    mel_spectrogram_dali_db = np.array(outputs[0][0])  # No need for .as_cpu()\n\n    return mel_spectrogram_dali_db\n    \n# ============================\n# 4. Tạo ảnh từ Mel Spectrogram\n# ============================\nfrom PIL import Image\nimport matplotlib.cm as cm\n\ndef get_spectrogram_image(mel_spec, cfg, output_size=(1280, 720)):\n    # Normalize mel spectrogram to 0–1\n    mel_spec = (mel_spec - mel_spec.min()) / (mel_spec.max() - mel_spec.min() + 1e-8)\n    # List of colormaps\n    cmap_name = cfg.COLOR_MAP\n    cmap = cm.get_cmap(cmap_name)\n    rgba_img = cmap(mel_spec)  # shape: (H, W, 4), values in [0, 1]\n    rgb_img = (rgba_img[:, :, :3] * 255).astype(np.uint8)\n    img = Image.fromarray(rgb_img)\n    img = img.transpose(Image.FLIP_TOP_BOTTOM)\n    img_resized = img.resize(output_size, Image.LANCZOS)\n    return img_resized\n\ndef get_and_save_spectrogram_image(mel_spec, cfg, output_dir, output_filename, output_size=(1280, 720)):\n    # List of colormaps\n    cmap_name = cfg.COLOR_MAP\n    out_path =  cfg.OUTPUT_FOLDER /cmap_name /output_dir\n    out_path.mkdir(parents=True, exist_ok=True)\n    output_file = out_path / output_filename\n    img_resized = get_spectrogram_image(mel_spec, cfg, output_size=output_size)\n    # Save final image\n    img_resized.save(output_file)\n    return img_resized, output_file\n    \ndef show_spectrogram(spec, title, sr, hop_length, y_axis=\"log\", x_axis=\"time\"):\n    librosa.display.specshow(\n        spec, sr=sr, y_axis=y_axis, x_axis=x_axis, hop_length=hop_length\n    )\n    plt.title(title)\n    plt.colorbar(format=\"%+2.0f dB\")\n    plt.tight_layout()\n    plt.show()\n\n# ============================\n # 5. Smoothing Label\n# ============================\n \ndef smooth_label(sub):\n    cols = sub.columns[1:]\n    groups = sub['row_id'].str.rsplit('_', n=1).str[0]\n    groups = groups.values\n    for group in np.unique(groups):\n        sub_group = sub[group == groups]\n        predictions = sub_group[cols].values\n        new_predictions = predictions.copy()\n        for i in range(1, predictions.shape[0]-1):\n            new_predictions[i] = (predictions[i-1] * 0.2) + (predictions[i] * 0.6) + (predictions[i+1] * 0.2)\n        new_predictions[0] = (predictions[0] * 0.9) + (predictions[1] * 0.1)\n        new_predictions[-1] = (predictions[-1] * 0.9) + (predictions[-2] * 0.1)\n        sub_group[cols] = new_predictions\n        sub[group == groups] = sub_group\n    return sub\n\n#==============================\n#6. Process 1 audio file\n#==============================\ndef process_audio_file(audio_path, model, output_folder, cfg, overlap_percent):\n    saved_data = [] # list of [row_id, saved image path] \n    samples_proc, sr = load_and_preprocess(audio_path, cfg.SAMPLE_RATE)\n    # Tạo sliding windows, mỗi window TARGET_DURATION_S giây, bước 0.5 giây\n    windows = get_sliding_windows(samples_proc, sr, cfg.TARGET_DURATION_S, overlap_percent = overlap_percent)\n    \n    # Lấy human voice intervals nếu tồn tại cho file này (dùng đường dẫn tương đối so với DATA_PATH)\n    for start_sample, end_sample in windows:\n        # Format for saving\n        window_start_time = start_sample / sr\n        base_name = os.path.splitext(Path(audio_path).name)[0]\n        output_filename = f\"{base_name}_{window_start_time:.1f}_{(window_start_time+cfg.TARGET_DURATION_S):.1f}.jpg\"\n        row_id = base_name + f'_{int (window_start_time+cfg.TARGET_DURATION_S)}'\n\n        chunk = samples_proc[start_sample:end_sample]\n        # Preprocessing \n        segment = np.array(chunk, dtype=np.float32)\n        mel_spec_db = compute_mel_spectrogram(segment, cfg)\n        pil_img, saved_path = get_and_save_spectrogram_image(mel_spec_db, cfg, output_folder, output_filename,(640,480))\n        x = data_transforms['test'](pil_img).unsqueeze(0) # (1, C, H, W)\n        # Model prediction\n        with torch.no_grad():\n            logits = model(x)  \n            scores = torch.sigmoid(logits)  \n            scores = scores.squeeze(0).cpu().numpy()  \n            \n        saved_data.append([row_id,saved_path]+ list(scores))\n    return saved_data   \n#==============================\n#6. Process 1 image file\n#==============================\ndef process_image_file(row_id, file_path, model, cfg):\n    pil_img = Image.open(file_path).convert('RGB')\n    x = data_transforms['test'](pil_img).unsqueeze(0).to(cfg.DEVICE) # (1, C, H, W)\n    # Model prediction\n    with torch.no_grad():\n        logits = model(x)  \n        scores = torch.sigmoid(logits)  \n        scores = scores.squeeze(0).cpu().numpy()  \n    return [[row_id,file_path]+ list(scores)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:17:09.713684Z","iopub.execute_input":"2025-05-03T05:17:09.714391Z","iopub.status.idle":"2025-05-03T05:17:09.733920Z","shell.execute_reply.started":"2025-05-03T05:17:09.714366Z","shell.execute_reply":"2025-05-03T05:17:09.733234Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from typing import Dict, Optional\nfrom transformers import (\n    AutoConfig,\n    ConvNextConfig,\n    ConvNextForImageClassification,\n    ConvNextModel,\n)\n\n#=============== Model Clas ========================\n\nclass ConvNextClassifier(nn.Module):\n    def __init__(\n        self,\n        num_channels: int = 1,\n        embedding_size: Optional[int] = None,\n    ):\n        super().__init__()\n        self.num_channels = num_channels\n        self.embedding_size = embedding_size\n        self.model = None\n        self.architecture = None\n        self._initialize_model()\n\n    def _initialize_model(self) -> nn.Module:\n        \"\"\"Initializes the ConvNext model based on specified attributes.\n\n        Returns:\n            nn.Module: The initialized ConvNext model.\n        \"\"\"\n\n        adjusted_state_dict = None\n\n        model = ConvNextModel\n        config = ConvNextConfig.from_pretrained(\"/kaggle/input/bc25-models-pt-files/convnext_base_facebook_224_22k\")\n        hidden_sizes = config.hidden_sizes\n        hidden_sizes[-1] = self.embedding_size\n        self.model = model.from_pretrained(\n                \"/kaggle/input/bc25-models-pt-files/convnext_base_facebook_224_22k\",\n                hidden_sizes=hidden_sizes,\n                num_channels=self.num_channels,\n                cache_dir=None,\n                ignore_mismatched_sizes=True,\n            )\n\n        self.architecture = self.model.config.model_type\n\n    def forward(self, input_values: torch.Tensor, labels: Optional[torch.Tensor] = None\n    ) -> torch.Tensor:\n\n        outputs = self.model(input_values)\n\n        output = outputs.last_hidden_state\n\n        return output\n        \nclass CustomNet(nn.Module):\n    def __init__(self, backbone, num_classes, add_on_layers_type=\"identity\", use_prototype = False, num_prototypes = 5):\n        super().__init__()\n        self.backbone = backbone\n        backbone_out = backbone.embedding_size\n        if add_on_layers_type == \"identity\":\n            self.add_on_layers = nn.Sequential(nn.Identity())\n        elif add_on_layers_type == \"upsample\":\n            self.add_on_layers = nn.Upsample(scale_factor=2, mode=\"bilinear\")\n        self.pooling = nn.AdaptiveAvgPool2d(1)\n        self.classifier = nn.Sequential(\n            self.add_on_layers,\n            nn.Linear(backbone_out, num_classes)\n        )\n        \n    def forward(self,x):\n        features = self.backbone(x)\n        if features.dim() == 4:\n            features = self.pooling(features)\n            features = features.view(features.size(0), -1)\n        \n        logits = self.classifier(features)\n        return logits","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:07:23.271672Z","iopub.execute_input":"2025-05-03T05:07:23.271863Z","iopub.status.idle":"2025-05-03T05:07:45.148321Z","shell.execute_reply.started":"2025-05-03T05:07:23.271848Z","shell.execute_reply":"2025-05-03T05:07:45.147715Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Submmission","metadata":{}},{"cell_type":"code","source":"# =====================Load model ===================\nbackbone = ConvNextClassifier(\n    num_channels=1,\n    embedding_size=1024,\n)\nmodel = CustomNet(backbone, cfg.NUM_CLASSES).to(cfg.DEVICE)\ncheckpoint = torch.load(cfg.MODEL_PATH, map_location=cfg.DEVICE)\n\nstate_dict_full= checkpoint['model_state_dict']\nprefix = '_orig_mod.'\nmodel_dict = {}\nfor key, val in state_dict_full.items():\n    new_key = key[len(prefix):]\n    model_dict[new_key] = val\nmodel.load_state_dict(model_dict, strict=False)\nmodel = model.to(cfg.DEVICE)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:13:43.044799Z","iopub.execute_input":"2025-05-03T05:13:43.045374Z","iopub.status.idle":"2025-05-03T05:13:44.847251Z","shell.execute_reply.started":"2025-05-03T05:13:43.045348Z","shell.execute_reply":"2025-05-03T05:13:44.846692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# not only visualize, but also for pre-allocated\nsig, rate = load_and_preprocess(\"/kaggle/input/birdclef-2025/train_soundscapes/H02_20230420_074000.ogg\", 32000)\nchunk = sig[32000*0:32000*5]\nsegment = np.array(chunk, dtype=np.float32)\nmel_spec_db = compute_mel_spectrogram(segment, cfg)\npil_img = get_spectrogram_image(mel_spec_db,cfg,(640,480))\ndisplay(pil_img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:07:54.132132Z","iopub.execute_input":"2025-05-03T05:07:54.132406Z","iopub.status.idle":"2025-05-03T05:08:07.097740Z","shell.execute_reply.started":"2025-05-03T05:07:54.132388Z","shell.execute_reply":"2025-05-03T05:08:07.097011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import time\n\n# Set seed\nnp.random.seed(cfg.SEED)\n# Prepare empty list for rows\nrows = []\n# sample submission:\nsubmission_example = pd.read_csv(cfg.SUBMISSION_SAMPLE_PATH)\n# Class labels from train audio\nclass_labels = sorted(cfg.CLASS_NAME)\n\nif cfg.USE_AUDIO_AS_INPUT:\n    # List of test soundscapes (only visible during submission)\n    test_soundscape_path = '/kaggle/input/birdclef-2025/train_soundscapes/'\n    test_soundscapes = [os.path.join(test_soundscape_path, afile) for afile in sorted(os.listdir(test_soundscape_path)) if afile.endswith('.ogg')]\n    \n    # Open each soundscape and make predictions for 5-second segments\n    # Use pandas df with 'row_id' plus class labels as columns\n\n    for count,soundscape in enumerate(test_soundscapes):\n        print(\"[\",count+1, \"/\", len(test_soundscapes),\"] processing file:\",soundscape)\n        rows += process_audio_file(audio_path= Path(soundscape),\n                              model = model,\n                              output_folder = cfg.OUTPUT_FOLDER,\n                              cfg = cfg,\n                              overlap_percent = 0.0,)\n\nelse: \n    metadata= pd.read_csv(cfg.MEATA_DATA_PATH)\n    old_prefix = \"/kaggle/working/train_soundscapes_data\"\n    new_prefix = str(cfg.DATA_PATH)\n    metadata['path'] = metadata['path'].str.replace(old_prefix, new_prefix, regex=False)\n    for idx, row in metadata.iterrows():\n        print(\"[\",idx+1, \"/\", len(metadata),\"] processing file:\",row['path'])\n        rows += process_image_file(row_id = row['row_id'], \n                                   file_path = row['path'],\n                                   model = model,\n                                   cfg = cfg)\n# Build predictions DataFrame\npredictions = pd.DataFrame(rows, columns=['row_id', 'path'] + class_labels)\n\n# Now remap to submission columns\nsubmission_cols = submission_example.columns[1:]\n\n# Reindex predictions to match submission\nsubmission = predictions[['row_id'] + list(submission_cols)]\nmeta = predictions[['row_id','path']]\n# Smoothing for final submission\nif cfg.USE_SMOOTH_LABEL:\n    submission = smooth_label(submission)\nsubmission.to_csv('submission.csv', index=False)\nmeta.to_csv('metadata_trainSoundscape.csv', index=False)\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:24:18.374468Z","iopub.execute_input":"2025-05-03T05:24:18.375159Z","iopub.status.idle":"2025-05-03T05:24:18.803642Z","shell.execute_reply.started":"2025-05-03T05:24:18.375135Z","shell.execute_reply":"2025-05-03T05:24:18.802839Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission.head(5)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:23:29.822629Z","iopub.execute_input":"2025-05-03T05:23:29.822910Z","iopub.status.idle":"2025-05-03T05:23:29.839352Z","shell.execute_reply.started":"2025-05-03T05:23:29.822890Z","shell.execute_reply":"2025-05-03T05:23:29.838595Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-03T05:23:31.303499Z","iopub.execute_input":"2025-05-03T05:23:31.303743Z","iopub.status.idle":"2025-05-03T05:23:31.310999Z","shell.execute_reply.started":"2025-05-03T05:23:31.303728Z","shell.execute_reply":"2025-05-03T05:23:31.310235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}