{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":142598,"sourceType":"datasetVersion","datasetId":3151},{"sourceId":11421803,"sourceType":"datasetVersion","datasetId":7153191},{"sourceId":11740995,"sourceType":"datasetVersion","datasetId":7370511},{"sourceId":11773242,"sourceType":"datasetVersion","datasetId":7212176}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# THIS NOTEBOOK USED FOR AUGMENTATION ONLY\n\nCreate a dataset included augmented data. Although it will increase the size of dataset (the original dataset size is around 8GB), it can help to reduce the bottle neck of training proces by reduce the pre-processing time. \n\nUse the [fast-audiomentation](https://github.com/Lallapallooza/fast-audiomentations/tree/main) for Adding noisy background to the dataset.","metadata":{}},{"cell_type":"code","source":"!pip install pyrootutils\n!pip install lightning\n!pip install torch-audiomentations\n!pip install noisereduce pedalboard\n!pip download --extra-index-url https://developer.download.nvidia.com/compute/redist/nightly nvidia-dali-nightly-cuda120\n!ls /kaggle/working |grep nvidia_dali_nightly_cuda120 |xargs pip install \n!rm -rf /kaggle/working/*\nimport sys\nfrom datasets import load_dataset\nimport os\nimport numpy as np\nimport pandas as pd\nimport pickle\nimport ast\nimport math\nimport time\nimport random\nimport gc\nimport cv2\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\nimport librosa\nimport matplotlib.pyplot as plt\n\nimport torch\nimport torchaudio\nfrom torchaudio import transforms\n\nimport IPython.display as ipd\nimport sys\nimport shutil\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, accuracy_score, confusion_matrix\n\nfrom nvidia.dali import pipeline_def\nimport nvidia.dali.fn as fn\nimport nvidia.dali.types as types\nimport nvidia.dali as dali\nfrom nvidia.dali.plugin.pytorch import DALIGenericIterator\nipd.clear_output()\nprint(\"Import finished\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-16T16:18:28.614231Z","iopub.execute_input":"2025-05-16T16:18:28.614884Z","iopub.status.idle":"2025-05-16T16:18:56.622325Z","shell.execute_reply.started":"2025-05-16T16:18:28.614857Z","shell.execute_reply":"2025-05-16T16:18:56.621509Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\nclass Config:\n    def __init__(self, **kwargs):\n        for k, v in kwargs.items():\n            setattr(self, k, v)\n\n    def update(self, **kwargs):\n        for k, v in kwargs.items():\n            setattr(self, k, v)\n\n# Initialize and set basic configuration\ncfg = Config(\n    SEED=42, \n    SAMPLE_RATE=32000,\n    META_DATA_PATH = Path(\"/kaggle/input/birdclef-2025/train.csv\"),\n    DATA_PATH=Path(\"/kaggle/input/birdclef-2025/train_audio\"),\n    NOISE_DATA_PATH=Path(\"/kaggle/working/no_call_augmented/noise\"),\n    OUTPUT_FOLDER =Path(\"/kaggle/working/256_2048/train_Mel_spec\"),\n    HUMAN_VOICE_PKL = Path('/kaggle/input/bc25-human-detect-sound/train_voice_data.pkl'),\n    MEL_COMBINATION = Path('/kaggle/input/data-augmentation-part4/mel_combination.csv'),\n    CLASS_NAME = np.load('/kaggle/input/metadate-bc25/class_names.npy', allow_pickle=True).tolist(),\n    NUM_CLASSES = 206,\n    CLASS2IDX={},\n    MEL_COMBINATION_INDEX = 16, # domain [0,15]\n    COLOR_MAP =['inferno'],\n    OUTPUT_METADATA=\"/kaggle/working/spec_img_meta.csv\",\n    WINDOW=\"hann\",\n    NFILTER_MEL=128,\n    WINDOW_LENGTH= 1024,\n    WINDOW_STEP= 512,\n    FREQ_HIGH=14000,\n    FREQ_LOW=150,\n    CUT_OFF_DB= -80,\n    TARGET_DURATION_S = 5,\n    TARGET_SAMPLES = 5*32000,\n    DEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\"),\n    BATCH_SIZE = 32,\n    NUM_WORKERS = 32,\n    )\ncfg.CLASS2IDX = {cls: idx for idx, cls in enumerate(cfg.CLASS_NAME)}\n# Function to seed everything to ensure reproducibility\ndef seed_everything(seed):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False # Change to true if input sizes are kept constant\n\nseed_everything(cfg.SEED)\n# Verifying changes\n# Device check\nprint(f\"Using device: {cfg.DEVICE}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T16:56:18.814487Z","iopub.execute_input":"2025-05-16T16:56:18.815144Z","iopub.status.idle":"2025-05-16T16:56:18.832277Z","shell.execute_reply.started":"2025-05-16T16:56:18.815119Z","shell.execute_reply":"2025-05-16T16:56:18.831513Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Utils ","metadata":{}},{"cell_type":"code","source":"# ============================\n# 1. Hàm load và tiền xử lý audio\n# ============================\ndef load_and_preprocess(audio_path, sr):\n    \"\"\"\n    - Load file .ogg bằng librosa.\n    \"\"\"\n    samples, sr = librosa.load(audio_path, sr=sr)\n    return samples, sr\n\n# ============================\n# 2. Hàm save Mel_specs thành file .npz\n# ============================\n\ndef save_mel_to_npz(mel_spectrogram: np.ndarray, output_path: str):\n    \"\"\"\n    Parameters:\n        mel_spectrogram (np.ndarray): 2D array of shape (n_mels, time_frames), dtype=float32/float64.\n        output_path (str): Path to save the .npz file (should end with .npz).\n    \"\"\"\n    if not isinstance(mel_spectrogram, np.ndarray):\n        raise ValueError(\"mel_spectrogram must be a NumPy array\")\n\n    if mel_spectrogram.ndim != 2:\n        raise ValueError(\"mel_spectrogram must be a 2D array (n_mels, time_frames)\")\n\n    # Convert to float16 if not already\n    mel_spectrogram = mel_spectrogram.astype(np.float16)\n\n    # Ensure directory exists\n    os.makedirs(os.path.dirname(output_path), exist_ok=True)\n\n    # Save to compressed .npz\n    np.savez_compressed(output_path, mel=mel_spectrogram)\n\n\n# ============================\n# 3. Hàm show Mel_spec\n# ============================\n\ndef load_and_plot_mel_npz(npz_path: Path, title: str = \"Mel Spectrogram\"):\n    \"\"\"\n    Load a Mel spectrogram from a .npz file and plot it.\n\n    Parameters:\n        npz_path (Path or str): Path to the .npz file.\n        title (str): Plot title.\n    \"\"\"\n    npz_path = Path(npz_path)\n\n    if not npz_path.exists():\n        raise FileNotFoundError(f\"File not found: {npz_path}\")\n    \n    data = np.load(npz_path)\n    if \"mel\" not in data:\n        raise KeyError(f\"'mel' key not found in {npz_path}\")\n    \n    mel = data[\"mel\"]\n\n    if mel.ndim != 2:\n        raise ValueError(\"Loaded mel spectrogram must be 2D\")\n\n    # Plot\n    plt.figure(figsize=(10, 4))\n    plt.imshow(mel, aspect='auto', origin='lower', cmap='magma')\n    plt.title(title)\n    plt.xlabel(\"Time Frames\")\n    plt.ylabel(\"Mel Bands\")\n    plt.colorbar(label=\"dB\")\n    plt.tight_layout()\n    plt.show()\n\n\n#=============================\n# 4. Metadata function handler\n# ====================\ndef to_idx_list(lbl_list: str, class2idx: dict[int,int]) -> list[int]:\n    \"\"\"\n    Convert a string representation of a list of labels (e.g. \"['123','456']\")\n    into a list of integer indices using class2idx mapping.\n    \"\"\"\n    labels = ast.literal_eval(lbl_list)\n    return [class2idx[l] for l in labels if l and l in class2idx]\n\ndef load_train_df(cfg, rating_threshold: float = 0.0) -> pd.DataFrame | None:\n    \"\"\"\n    Load metadata CSV, filter by rating, and prepare:\n      - 'label' (int)\n      - 'secondary_label_idx' (list[int])\n      - 'ogg_path' (str)\n      - 'npz_path' (str)\n    \"\"\"\n    meta_path = Path(cfg.META_DATA_PATH)\n    if not meta_path.exists():\n        return None\n\n    # 1) Load and filter\n    df = pd.read_csv(meta_path)\n    train_df = df[df['rating'] > rating_threshold].copy()\n\n    # 2) Drop unused cols (if present)\n    drop_cols = [\n        'url','license','common_name','collection','author',\n        'type','latitude','longitude','scientific_name'\n    ]\n    train_df.drop(columns=[c for c in drop_cols if c in train_df], inplace=True)\n\n    # 3) Primary label → integer\n    train_df['label'] = train_df['primary_label'].map(cfg.CLASS2IDX)\n\n    # 4) Secondary labels → list of ints\n    train_df['secondary_label_idx'] = train_df['secondary_labels']\\\n        .apply(lambda s: to_idx_list(s, cfg.CLASS2IDX))\n\n    # 5) /kaggle/input/birdclef-2025/train_audio/XXX.ogg\n    train_df['ogg_path'] = train_df['filename'].apply(\n        lambda fn: str(Path(\"/kaggle/input/birdclef-2025/train_audio\") / fn)\n    )\n\n    # 7) Summary\n    print(\"\\nColumns:\", train_df.columns.tolist())\n    print(\"Total species:\", df['primary_label'].nunique())\n    print(\"Filtered species:\", train_df['primary_label'].nunique())\n    print(\"Secondary-label counts:\", train_df['secondary_labels'].nunique())\n    print(\"Top-10 distribution:\\n\", train_df['primary_label'].value_counts().head(10))\n\n    return train_df\n    \ndef load_human_voice(pkl_path):\n    with open(pkl_path, \"rb\") as f:\n        human_voice_dict = pickle.load(f)\n    return human_voice_dict","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T16:56:20.256099Z","iopub.execute_input":"2025-05-16T16:56:20.256378Z","iopub.status.idle":"2025-05-16T16:56:20.268451Z","shell.execute_reply.started":"2025-05-16T16:56:20.256358Z","shell.execute_reply":"2025-05-16T16:56:20.267841Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Create Augmentation dataset","metadata":{}},{"cell_type":"markdown","source":"By taking the advantages of the ESC50 dataset, which is the environmental sound classification dataset, I dropped all categories related to animals or too noisy sound ( fireworks, ...), the left categories will be used as the background noise.","metadata":{}},{"cell_type":"markdown","source":"The truth is that not every 5s audio segment cut from train_audio has bird/frog sounds, so it is certain that there will be false negatives.\n\n**In this section, we will do 5 steps for data preprocessing:**\n\n1. Remove human voice in the human_dict_pd and silence segment from the audio\n2. Create a no_call_augmented folder containing unrelevant sound to birds, frog and animals.\n3. Create augmented_train_audio by adding background noise.\n4. Convert to mel-spectrogram then save the mel as files .npz\n5. Run with the size as large as posible ( run until the notebook broken ).","metadata":{}},{"cell_type":"code","source":"\n#=========================================\n# STEP 1 Remove human voice and silence segment  \n#=========================================\n\nfrom pydub import AudioSegment\nfrom pydub.silence import split_on_silence\nfrom torch.utils.data import Dataset, DataLoader\n\n# ======== PROCESS FUNCTION ========\n    \ndef remove_human_voice(samples, sr, human_voice_intervals, pad_duration=0.5, save_human_call_data=False):\n    \"\"\"\n    Cắt bỏ các đoạn audio trùng với tiếng người và nối lại các đoạn còn lại.\n    THAY ĐỔI SO VỚI BAN ĐẦU: thêm offset 0.5s ở start và end time.\n    Args:\n        samples (np.ndarray): mảng 1 chiều waveform.\n        sr (int): tần số lấy mẫu (sample rate).\n        human_voice_intervals (list of dict): [{'start': float, 'end': float}], đơn vị giây.\n        pad_duration (float): thời gian (giây) chèn padding (im lặng) giữa các đoạn.\n        save_human_call_data (bool): nếu True, sẽ lưu các đoạn chứa tiếng người vào folder `human_call_data`.\n\n    Returns:\n        np.ndarray: mảng audio mới đã loại bỏ tiếng người.\n    \"\"\"\n    total_duration = len(samples) / sr\n    result = []\n\n    # Sort intervals by start time\n    intervals = sorted(human_voice_intervals, key=lambda x: x['start'])\n    current_pos = 0\n\n    pad_samples = int(pad_duration * sr)\n    silence_pad = np.zeros(pad_samples, dtype=samples.dtype)\n\n    # Prepare folder to save clips of human voice if needed\n    if save_human_call_data:\n        out_dir = Path(\"human_call_data\")\n        out_dir.mkdir(parents=True, exist_ok=True)\n        timestamp = datetime.now().strftime(\"%Y%m%d_%H%M%S\")\n\n    for idx, interval in enumerate(intervals):\n        start_sample = int(max(0,interval['start'] - 0.5) * sr) #offset 0.5 s\n        end_sample = int(min(total_duration, interval['end'] + 0.5) * sr)  #offset 0.5 s\n\n        # Lưu đoạn có tiếng người\n        if save_human_call_data:\n            segment = samples[start_sample:end_sample]\n            filename = out_dir / f\"human_{timestamp}_part{idx}_from_{interval['start']:.2f}s_to_{interval['end']:.2f}s.wav\"\n            sf.write(filename, segment, sr)\n\n        # Lấy phần trước đoạn tiếng người\n        if start_sample > current_pos:\n            segment = samples[current_pos:start_sample]\n            result.append(segment)\n            if pad_duration > 0:\n                result.append(silence_pad)\n\n        current_pos = max(current_pos, end_sample)\n\n    # Lấy phần cuối sau đoạn tiếng người\n    if current_pos < len(samples):\n        result.append(samples[current_pos:])\n\n    if result:\n        return np.concatenate(result)\n    else:\n        return np.zeros(1, dtype=samples.dtype)  # fallback: toàn bộ là tiếng người\n    \nfrom pydub import AudioSegment\nfrom pydub.silence import split_on_silence\n\ndef remove_silience(samples, sr):\n    \"\"\"\n    Cắt bỏ những phần âm thanh yên lặng có độ dài lớn hơn 1 giây\n    \"\"\"\n    try:\n        # Convert to int16 for AudioSegment\n        samples_int16 = (samples * 32767).astype(np.int16)\n\n        audio = AudioSegment(\n            samples_int16.tobytes(),\n            frame_rate=sr,\n            sample_width=2,  # int16 = 2 bytes\n            channels=1\n        )\n\n        # Split on silence\n        audio_chunks = split_on_silence(audio,\n                                        min_silence_len=1000,\n                                        silence_thresh=-45,\n                                        keep_silence=100)\n\n        # Reconstruct\n        if not audio_chunks:\n            combined = audio  # fallback\n        else:\n            combined = AudioSegment.empty()\n            for chunk in audio_chunks:\n                combined += chunk\n\n        out_samples = np.array(combined.get_array_of_samples()).astype(np.float32)\n        return out_samples / 32767.0\n\n    except Exception as e:\n        print(f\"❌ Error in remove_silience: {e}\")\n        return samples  # fallback: trả lại input gốc\n\n\n#=========================================\n# STEP 2 no_call_augmented folder containing \n#=========================================\ndef save_esc50_dataset(saved_path):\n    # Sẽ có tổng cộng 800 file audio 5s liên quan tới âm thanh về thiên nhiên hoặc đồ vật được lưu lại\n    # 1) Đọc metadata và định nghĩa danh sách các category cần augment\n    esc50 = pd.read_csv(\"/kaggle/input/environmental-sound-classification-50/esc50.csv\")\n    augmented_category = [\n        'chainsaw', 'vacuum_cleaner', 'door_wood_knock', 'can_opening', 'crow', 'clapping',\n        'pouring_water', 'water_drops', 'church_bells', 'keyboard_typing', 'wind', 'footsteps',\n        'brushing_teeth', 'crackling_fire', 'drinking_sipping', 'snoring', 'washing_machine',\n        'clock_tick', 'door_wood_creaks', 'sea_waves'\n    ]\n    \n    # 2) Thư mục gốc chứa file .wav\n    src_root = Path(\"/kaggle/input/environmental-sound-classification-50/audio/audio/44100\")\n    \n    # 3) Thư mục đích để lưu các file đã chọn\n    dest_root = Path(\"/kaggle/working/no_call_augmented/noise\")\n    dest_root.mkdir(parents=True, exist_ok=True)\n    \n    # 4) Lọc DataFrame\n    df_sel = esc50[esc50[\"category\"].isin(augmented_category)]\n    \n    # 5) Copy từng file\n    for _, row in df_sel.iterrows():\n        filename = row[\"filename\"]          # ví dụ \"1-100032-A-0.wav\"\n        cat      = row[\"category\"]\n        \n        src_path  = src_root / filename\n        dest_path = dest_root / filename\n    \n        # copy file\n        if src_path.exists():\n            shutil.copy(src_path, dest_path)\n        else:\n            print(f\"WARNING: không tìm thấy {src_path}\")\n    \n    print(\"Done saving unrelevant data from esc50 process!\")\n\n#===================================\n# STEP 3 AUGMENTATION \n#====================================\nfrom torch_audiomentations import Compose, Gain, PolarityInversion, AddBackgroundNoise, AddColoredNoise\n\ndef build_torch_audio_augment_pipeline(\n    background_noise_path,\n    sample_rate=32000,\n    min_snr_db=5,\n    max_snr_db=20\n):\n    augment = Compose(\n        transforms=[\n            Gain(min_gain_in_db=-6.0, max_gain_in_db=6.0, p=0.5, output_type=\"tensor\"),\n            PolarityInversion(p=0.3, output_type=\"tensor\"),\n            AddBackgroundNoise(\n                background_paths=background_noise_path,\n                sample_rate=sample_rate,\n                min_snr_in_db=min_snr_db,\n                max_snr_in_db=max_snr_db,\n                p=0.7,\n                output_type=\"tensor\"\n            ),\n            AddColoredNoise(\n                min_snr_in_db=min_snr_db,\n                max_snr_in_db=max_snr_db,\n                min_f_decay=-2.0,\n                max_f_decay=2.0,\n                sample_rate=sample_rate,\n                p=0.5,\n                output_type=\"tensor\"\n            )\n        ],\n        output_type=\"tensor\"  # Cấu hình cho toàn bộ pipeline\n    )\n    print(\"Finished build augmentation pipeline\")\n    return augment\n\n# ============================\n# STEP 4. Hàm Tạo Mel Spectrogram\n# ============================\nimport torchaudio\n\n# Khởi tạo bộ chuyển đổi (chỉ cần tạo 1 lần)\n\ndef compute_mel_spectrogram_pytorch(cfg):\n    mel_transform = torchaudio.transforms.MelSpectrogram(\n        sample_rate=cfg.SAMPLE_RATE,\n        n_fft=cfg.WINDOW_LENGTH,\n        win_length=cfg.WINDOW_LENGTH,\n        hop_length=cfg.WINDOW_STEP,\n        n_mels=cfg.NFILTER_MEL,\n        f_min=cfg.FREQ_LOW,\n        f_max=cfg.FREQ_HIGH,\n        power = 2.0,\n        \n    ).to(cfg.DEVICE)\n    \n    db_transform = torchaudio.transforms.AmplitudeToDB(stype=\"power\", top_db=-cfg.CUT_OFF_DB).to(cfg.DEVICE)\n    print(\"Finished build mel spectrograme transform pipeline\")\n    return mel_transform, db_transform  \n\n#==============================\n# STEP 5: Full pipeline\n#==============================\nclass MyDataset(Dataset):\n    def __init__(self, \n                 meta_df, \n                 human_voice_dict,\n                 cfg,\n                 transforms=None,\n                 give_label=False):\n        \"\"\"\n        Args:\n            meta_df (pd.DataFrame): DataFrame chứa ít nhất cột \"path\" (đường dẫn ảnh).\n            transforms: torchvision transforms sẽ được áp dụng.\n            give_label (bool): True nếu có label (dùng cho train/val), False nếu dùng cho test.\n        \"\"\"\n        super().__init__()\n        self.df = meta_df.reset_index(drop=True).copy()\n        self.hv = human_voice_dict\n        self.sr = cfg.SAMPLE_RATE\n        self.hop_length = cfg.WINDOW_LENGTH//2\n        self.give_label = give_label\n        self.seg_frames = int(5.0 * self.sr / self.hop_length)\n        self.num_classes = cfg.NUM_CLASSES\n        self.labels = self.df['label'].values\n\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        target = self.labels[index]\n        ogg_path   = self.df.loc[index, \"ogg_path\"]\n        \n        mel_spec   = load_mel_spec(npz_path)            # shape (F, T)\n        human_iv   = self.hv.get(ogg_path, [])          # list of {'start','end'}\n\n        chunk    = get_chunks(mel_spec.shape[1], human_iv, self.sr, self.hop_length, self.seg_frames)        \n        seg       = mel_spec[:, chunk[\"start\"]:chunk[\"end\"]]  # still (F, seg_frames)\n        # normalize to [0,1]\n        seg = (seg - seg.min()) / (seg.max() - seg.min() + 1e-8)\n        \n        # pad zeros on the right\n        L = seg.shape[1]\n        if L < self.seg_frames:\n            pad_width = self.seg_frames - L\n            seg = np.pad(seg,\n                         pad_width=((0,0), (0, pad_width)),\n                         mode='constant',\n                         constant_values=0)\n        \n        # Áp dụng transform\n        if self.transform:\n            pil = to_pil_image(seg, mode='F')\n            x = self.transform(pil)        # now your Compose([ToTensor(),Resize,...]) works\n        else:\n            x = torch.from_numpy(seg).unsqueeze(0).float()\n        \n        return (x, target) if self.give_label else x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T18:06:08.610742Z","iopub.execute_input":"2025-05-16T18:06:08.611376Z","iopub.status.idle":"2025-05-16T18:06:08.632699Z","shell.execute_reply.started":"2025-05-16T18:06:08.611350Z","shell.execute_reply":"2025-05-16T18:06:08.631856Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport torchaudio\nfrom pathlib import Path\nfrom tqdm import tqdm\n# FULL pipeline function\ndef augment_and_save_chunks(\n    noise_dir,\n    out_dir,\n    sr,\n    segment_seconds,\n    augment_pipeline,\n    mel_transform,\n    db_transform,\n):\n    out_dir = Path(out_dir)\n    out_dir.mkdir(parents=True, exist_ok=True)\n\n    segment_len = int(segment_seconds * sr)\n    audio_paths = list(Path(noise_dir).rglob(\"*.wav\"))\n    \n    for path in tqdm(audio_paths, desc=\"Processing\"):\n        waveform, _ = torchaudio.load(path)\n        waveform = waveform.mean(dim=0, keepdim=True)\n        waveform = torchaudio.functional.resample(waveform, orig_freq=44100, new_freq=sr)\n        waveform = waveform / waveform.abs().max()\n\n        with torch.no_grad():\n            augmented = augment_pipeline(waveform)\n\n        audio_np = augmented.squeeze().cpu().numpy()\n\n        # Split thành các đoạn nhỏ\n        total_samples = len(audio_np)\n        for i in range(0, total_samples - segment_len + 1, segment_len):\n            chunk = audio_np[i:i+segment_len]\n            chunk_tensor = torch.tensor(chunk, dtype=torch.float32).unsqueeze(0)\n\n            mel = mel_transform(chunk_tensor)\n            mel_db = db_transform(mel).squeeze().cpu().numpy()\n\n            mel_db = mel_db.astype(np.float16)\n\n            # Lưu file\n            out_path = out_dir / f\"{path.stem}_chunk{i}.npz\"\n            np.savez_compressed(out_path, mel=mel_db)\n\n    print(f\"✅ Done saving augmented spectrograms to: {out_dir}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T17:25:39.886402Z","iopub.execute_input":"2025-05-16T17:25:39.887145Z","iopub.status.idle":"2025-05-16T17:25:39.893799Z","shell.execute_reply.started":"2025-05-16T17:25:39.887120Z","shell.execute_reply":"2025-05-16T17:25:39.893238Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Testing augmentation with a random audio","metadata":{}},{"cell_type":"code","source":"df = load_train_df(cfg)\nhv = load_human_voice(cfg.HUMAN_VOICE_PKL)\nsave_esc50_dataset(cfg.NOISE_DATA_PATH)\naugment_transform = build_torch_audio_augment_pipeline(background_noise_path = str(cfg.NOISE_DATA_PATH), \n                                                        sample_rate=32000,\n                                                        min_snr_db=2,\n                                                        max_snr_db=15)\nmel_transform, db_transform = compute_mel_spectrogram_pytorch(cfg)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T17:37:14.608715Z","iopub.execute_input":"2025-05-16T17:37:14.609484Z","iopub.status.idle":"2025-05-16T17:37:16.330687Z","shell.execute_reply.started":"2025-05-16T17:37:14.609458Z","shell.execute_reply":"2025-05-16T17:37:16.329916Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ====== Remove human voice and silence segment  =============\nfpath   = \"/kaggle/input/birdclef-2025/train_audio/1139490/CSA36389.ogg\"\nhuman_iv   = hv.get(fpath, [])          # list of {'start','end'}\n# Load raw waveform\nwaveform, sr = torchaudio.load(fpath)\nwaveform = waveform.mean(dim=0, keepdim=True)  # mono\nwaveform = torchaudio.functional.resample(waveform, orig_freq=sr, new_freq=cfg.SAMPLE_RATE)\n\n# Normalize [-1,1]\nwaveform = waveform / waveform.abs().max()\nsamples = waveform.squeeze().cpu().numpy()\n\n# waveform = waveform.reshape(1,-1).unsqueeze(0)\n\n# Remove human noise \nsamples_no_human = remove_human_voice(samples, sr, human_iv, pad_duration=0.5, save_human_call_data=False)\n# Remove silence segment \nsamples_final = remove_silience(samples_no_human, sr)\n\n# Plot\nfig, ax = plt.subplots(3, 1, figsize=(12, 4))\nax[0].plot(samples)\nax[0].set_title(\"Original waveform\")\nax[1].plot(samples_no_human)\nax[1].set_title(\"Removed human waveform\")\nax[2].plot(samples_final)\nax[2].set_title(\"Removed human and silence waveform\")\nplt.tight_layout()\nplt.show()\n\n# Nghe thử\nprint(\"🔊 Original\")\nipd.display(ipd.Audio(samples, rate=sr))\nprint(\"🔊 Removed human\")\nipd.display(ipd.Audio(samples_no_human, rate=sr))\nprint(\"🔊 Removed human and silence\")\nipd.display(ipd.Audio(samples_final, rate=sr))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T18:06:58.432054Z","iopub.execute_input":"2025-05-16T18:06:58.432355Z","iopub.status.idle":"2025-05-16T18:07:00.434424Z","shell.execute_reply.started":"2025-05-16T18:06:58.432333Z","shell.execute_reply":"2025-05-16T18:07:00.433783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ======= Test augmentation code ========== \nfpath = '/kaggle/input/birdclef-2025/train_audio/22333/XC890507.ogg'\n# Load raw waveform\nwaveform, sr = torchaudio.load(fpath)\nwaveform = waveform.mean(dim=0, keepdim=True)  # mono\nwaveform = torchaudio.functional.resample(waveform, orig_freq=sr, new_freq=cfg.SAMPLE_RATE)\n\n# Normalize [-1,1]\nwaveform = waveform / waveform.abs().max()\nwaveform = waveform.reshape(1,-1).unsqueeze(0)\n\n# Apply augment\nwith torch.no_grad():\n    augmented = augment_transform(waveform)\n\n# Convert to numpy\nraw_np = waveform.squeeze().cpu().numpy()\naug_np = augmented.squeeze().cpu().numpy()\n\n# Plot\nfig, ax = plt.subplots(2, 1, figsize=(12, 4))\nax[0].plot(raw_np)\nax[0].set_title(\"Original waveform\")\nax[1].plot(aug_np)\nax[1].set_title(\"Augmented waveform\")\nplt.tight_layout()\nplt.show()\n\n# Nghe thử\nprint(\"🔊 Original\")\nipd.display(ipd.Audio(raw_np, rate=sr))\nprint(\"🔊 Augmented\")\nipd.display(ipd.Audio(aug_np, rate=sr))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T17:42:19.939862Z","iopub.execute_input":"2025-05-16T17:42:19.940500Z","iopub.status.idle":"2025-05-16T17:42:20.649683Z","shell.execute_reply.started":"2025-05-16T17:42:19.940476Z","shell.execute_reply":"2025-05-16T17:42:20.649064Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}