{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11977319,"sourceType":"datasetVersion","datasetId":7212176},{"sourceId":11421803,"sourceType":"datasetVersion","datasetId":7153191}],"dockerImageVersionId":31041,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pyrootutils\n# !pip install lightning\n!pip install torch-audiomentations\n!pip install noisereduce pedalboard\n!pip install --extra-index-url https://pypi.nvidia.com --upgrade nvidia-dali-cuda120","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-29T14:54:41.443849Z","iopub.execute_input":"2025-05-29T14:54:41.444049Z","iopub.status.idle":"2025-05-29T14:56:30.484597Z","shell.execute_reply.started":"2025-05-29T14:54:41.444022Z","shell.execute_reply":"2025-05-29T14:56:30.483831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from nvidia.dali.pipeline import pipeline_def\nimport nvidia.dali.fn as fn\nimport nvidia.dali.types as types\nfrom nvidia.dali.plugin.pytorch import DALIGenericIterator\nfrom pathlib import Path\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport librosa\nimport os\nimport torch\nimport pandas as pd\nimport pickle\nfrom typing import Union\nimport ast\nimport soundfile as sf\nfrom datetime import datetime\nimport random\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-29T15:03:12.928950Z","iopub.execute_input":"2025-05-29T15:03:12.929239Z","iopub.status.idle":"2025-05-29T15:03:12.934538Z","shell.execute_reply.started":"2025-05-29T15:03:12.929217Z","shell.execute_reply":"2025-05-29T15:03:12.933724Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    def __init__(self, **kwargs):\n        for k, v in kwargs.items():\n            setattr(self, k, v)\n\n    def update(self, **kwargs):\n        for k, v in kwargs.items():\n            setattr(self, k, v)\n\n# Initialize and set basic configuration\n\ncfg = Config(\n    SEED=42, \n    SAMPLE_RATE=32000,\n    META_DATA_PATH = Path(\"/kaggle/input/birdclef-2025/train.csv\"),\n    DATA_PATH=Path(\"/kaggle/input/birdclef-2025/train_audio\"),\n    OUTPUT_FOLDER =Path(\"train_Mel_spec\"),\n    HUMAN_VOICE_PKL = Path('/kaggle/input/bc25-human-detect-sound/train_voice_data.pkl'),\n    CLASS_NAME = np.load('/kaggle/input/metadate-bc25/class_names.npy', allow_pickle=True).tolist(),\n    NUM_CLASSES = 206,\n    CLASS2IDX={},\n    MEL_COMBINATION_INDEX = 16, # domain [0,15]\n    COLOR_MAP =['inferno'],\n    OUTPUT_METADATA=\"spec_img_meta.csv\",\n    WINDOW=\"hann\",\n    NFILTER_MEL=256,\n    WINDOW_LENGTH= 2048,\n    WINDOW_STEP= 1024,\n    FREQ_HIGH=14000,\n    FREQ_LOW=100,\n    CUT_OFF_DB= -80,\n    TARGET_DURATION_S = 5,\n    TARGET_SAMPLES = 5*32000,\n    DEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\"),\n    BATCH_SIZE = 32,\n    NUM_WORKERS = 32,\n    )\ncfg.CLASS2IDX = {cls: idx for idx, cls in enumerate(cfg.CLASS_NAME)}\n\n# Function to seed everything to ensure reproducibility\ndef seed_everything(seed):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False # Change to true if input sizes are kept constant\n    \nseed_everything(cfg.SEED)\nprint(f\"Using device: {cfg.DEVICE}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-29T15:05:52.530006Z","iopub.execute_input":"2025-05-29T15:05:52.530504Z","iopub.status.idle":"2025-05-29T15:05:52.541194Z","shell.execute_reply.started":"2025-05-29T15:05:52.530481Z","shell.execute_reply":"2025-05-29T15:05:52.540661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\n\n#=============================\n# 4. Metadata function handler\n# ====================\nfrom typing import Dict, List\n\ndef to_idx_list(lbl_list: str, class2idx: Dict[int, int]) -> List[int]:\n    \"\"\"\n    Convert a string representation of a list of labels (e.g. \"['123','456']\")\n    into a list of integer indices using class2idx mapping.\n    \"\"\"\n    labels = ast.literal_eval(lbl_list)\n    return [class2idx[l] for l in labels if l and l in class2idx]\n\n\n\ndef load_train_df(cfg, rating_threshold: float = 0.0) -> Union[pd.DataFrame, None]:\n\n    \"\"\"\n    Load metadata CSV, filter by rating, and prepare:\n      - 'label' (int)\n      - 'secondary_label_idx' (list[int])\n      - 'ogg_path' (str)\n      - 'npz_path' (str)\n    \"\"\"\n    meta_path = Path(cfg.META_DATA_PATH)\n    if not meta_path.exists():\n        return None\n\n    # 1) Load and filter\n    df = pd.read_csv(meta_path)\n    train_df = df[df['rating'] >= rating_threshold].copy()\n\n    # 2) Drop unused cols (if present)\n    drop_cols = [\n        'url','license','common_name','collection','author',\n        'type','latitude','longitude','scientific_name'\n    ]\n    train_df.drop(columns=[c for c in drop_cols if c in train_df], inplace=True)\n\n    # 3) Primary label → integer\n    train_df['label'] = train_df['primary_label'].map(cfg.CLASS2IDX)\n\n    # 4) Secondary labels → list of ints\n    train_df['secondary_label_idx'] = train_df['secondary_labels']\\\n        .apply(lambda s: to_idx_list(s, cfg.CLASS2IDX))\n\n    # 5) /kaggle/input/birdclef-2025/train_audio/XXX.ogg\n    train_df['ogg_path'] = train_df['filename'].apply(\n        lambda fn: str(cfg.DATA_PATH / fn)\n    )\n    #6) drop non-existing files\n    # train_df = train_df[train_df['ogg_path'].apply(lambda x: Path(x).exists())]\n    # 7) Summary\n    print(\"\\nColumns:\", train_df.columns.tolist())\n    print(\"Total species:\", df['primary_label'].nunique())\n    print(\"Filtered species:\", train_df['primary_label'].nunique())\n    print(\"Secondary-label counts:\", train_df['secondary_labels'].nunique())\n    print(\"Top-10 distribution:\\n\", train_df['primary_label'].value_counts().head(10))\n\n    return train_df\n    \ndef load_human_voice(pkl_path):\n    with open(pkl_path, \"rb\") as f:\n        human_voice_dict = pickle.load(f)\n    return human_voice_dict\n\n\n\n\n\n# ======== PROCESS FUNCTION ========\n    \ndef remove_human_voice(samples, sr, human_voice_intervals, pad_duration=0.5, save_human_call_data=False):\n    \"\"\"\n    Xóa bỏ tiếng người khỏi tín hiệu âm thanh, hoạt động với cả NumPy và PyTorch.\n\n    Args:\n        samples (np.ndarray or torch.Tensor): waveform 1 chiều.\n        sr (int): sample rate.\n        human_voice_intervals (list of dict): [{'start': float, 'end': float}], đơn vị giây.\n        pad_duration (float): thời gian im lặng chèn giữa các đoạn.\n        save_human_call_data (bool): lưu tiếng người nếu True.\n\n    Returns:\n        cùng kiểu với `samples`: audio đã loại tiếng người.\n    \"\"\"\n    is_torch = isinstance(samples, torch.Tensor)\n    if is_torch:\n        samples_np = samples.cpu().numpy()\n    else:\n        samples_np = samples\n\n    total_duration = len(samples_np) / sr\n    result = []\n\n    # Sort intervals\n    intervals = sorted(human_voice_intervals, key=lambda x: x['start'])\n    current_pos = 0\n    pad_samples = int(pad_duration * sr)\n    silence_pad = np.zeros(pad_samples, dtype=samples_np.dtype)\n\n    # Tạo folder lưu nếu cần\n    if save_human_call_data:\n        out_dir = Path(\"human_call_data\")\n        out_dir.mkdir(parents=True, exist_ok=True)\n        timestamp = datetime.now().strftime(\"%Y%m%d_%H%M%S\")\n\n    for idx, interval in enumerate(intervals):\n        start_sample = int(max(0, interval['start'] - 0.5) * sr)\n        end_sample = int(min(total_duration, interval['end'] + 0.5) * sr)\n\n        if save_human_call_data:\n            segment = samples_np[start_sample:end_sample]\n            filename = out_dir / f\"human_{timestamp}_part{idx}_from_{interval['start']:.2f}s_to_{interval['end']:.2f}s.wav\"\n            sf.write(filename, segment, sr)\n\n        if start_sample > current_pos:\n            result.append(samples_np[current_pos:start_sample])\n            if pad_duration > 0:\n                result.append(silence_pad)\n        current_pos = max(current_pos, end_sample)\n\n    if current_pos < len(samples_np):\n        result.append(samples_np[current_pos:])\n\n    if result:\n        output_np = np.concatenate(result)\n        if len(output_np) > sr:\n            return torch.from_numpy(output_np) if is_torch else output_np\n\n    return samples  # giữ nguyên nếu toàn bộ là tiếng người\n\ndef remove_silence(samples, sr, silence_thresh_db=-45.0, min_silence_len=1.0, keep_silence=0.1):\n    \"\"\"\n    Xóa các đoạn yên lặng dài hơn `min_silence_len` giây.\n    Hỗ trợ cả numpy và PyTorch.\n\n    Args:\n        samples (np.ndarray or torch.Tensor): waveform 1 chiều.\n        sr (int): sample rate.\n        silence_thresh_db (float): ngưỡng dBFS để xem là yên lặng.\n        min_silence_len (float): độ dài tối thiểu (giây) để tính là yên lặng.\n        keep_silence (float): thời lượng (giây) được giữ lại ở đầu/cuối mỗi đoạn không yên lặng.\n\n    Returns:\n        cùng kiểu với `samples`: audio đã loại đoạn yên lặng.\n    \"\"\"\n    is_torch = isinstance(samples, torch.Tensor)\n    if is_torch:\n        samples_np = samples.cpu().numpy()\n    else:\n        samples_np = samples\n\n    samples_np = samples_np.astype(np.float32)\n    win_size = int(0.02 * sr)  # 20ms window\n    hop_size = int(0.01 * sr)  # 10ms hop\n    min_silence_samples = int(min_silence_len * sr)\n    keep_samples = int(keep_silence * sr)\n\n    # Compute short-time RMS\n    rms = np.array([\n        np.sqrt(np.mean(samples_np[i:i + win_size] ** 2))\n        for i in range(0, len(samples_np) - win_size + 1, hop_size)\n    ])\n\n    # Convert to dBa\n    eps = 1e-10\n    rms_db = 20 * np.log10(rms + eps)\n    silence_mask = rms_db < silence_thresh_db\n\n    # Expand frame-wise mask to sample-wise\n    frame_mask = np.repeat(silence_mask, hop_size)\n    frame_mask = np.pad(frame_mask, (0, len(samples_np) - len(frame_mask)), constant_values=True)\n\n    # Invert to find non-silent regions\n    non_silent_mask = ~frame_mask\n    change_points = np.diff(non_silent_mask.astype(np.int8), prepend=0)\n    start_indices = np.where(change_points == 1)[0]\n    end_indices = np.where(change_points == -1)[0]\n\n    if len(end_indices) < len(start_indices):\n        end_indices = np.append(end_indices, len(samples_np))\n\n    result = []\n    for start, end in zip(start_indices, end_indices):\n        if end - start >= min_silence_samples:\n            s = max(0, start - keep_samples)\n            e = min(len(samples_np), end + keep_samples)\n            result.append(samples_np[s:e])\n\n    if result:\n        output_np = np.concatenate(result)\n        if len(output_np) > sr:\n            return torch.from_numpy(output_np) if is_torch else output_np\n\n    return samples  # fallback: giữ nguyên nếu toàn bộ là yên lặng\n\n#=========================\n# 5. DALI pipeline\n# =========================\n\nclass ExternalInputIterator:\n    def __init__(self, df, cfg, reduce_human_voice=True, reduce_silence=True, shuffle =False):\n        self.reduce_human_voice = reduce_human_voice\n        self.reduce_silence = reduce_silence\n        self.batch_size = cfg.BATCH_SIZE\n        self.sample_rate = cfg.SAMPLE_RATE\n        self.human_voice_dict = load_human_voice(cfg.HUMAN_VOICE_PKL)\n        self.df = df\n        # if shuffle:\n        #     self.df = self.df.sample(frac=1).reset_index(drop=True)\n        self.i = 0\n        self.n = len(self.df)\n\n    def __iter__(self):\n        self.i = 0\n        return self\n\n    def __next__(self):\n        if self.i >= self.n:\n            self.i = 0\n            raise StopIteration\n        batch_audio = []\n        batch_labels = []\n        batch_idx = []\n        # self.current_filenames = []\n\n        for _ in range(self.batch_size):\n            if self.i >= self.n:\n                break\n            row = self.df.iloc[self.i]\n            audio_path = row['ogg_path']\n            label = row['label']\n            print(f\"Processing {self.i+1}/{self.n}: {audio_path}\")\n            samples, sr = librosa.load(audio_path, sr=self.sample_rate)\n\n            if self.batch_size == 1: # batch size 1 dùng cho tạo dataset\n                if len(samples) > 600 * self.sample_rate: # Giới hạn độ dài tối đa là 600s\n                    samples = samples[:600 * self.sample_rate]\n\n            if self.reduce_human_voice:\n                remap_path = Path('/kaggle/input/birdclef-2025/train_audio/')/row['filename']\n                human_voice_intervals = self.human_voice_dict.get(str(remap_path), [])\n                if human_voice_intervals:\n                    samples = remove_human_voice(samples, sr, human_voice_intervals)\n            \n            if self.reduce_silence:\n                samples = remove_silence(samples, sr, silence_thresh_db=-45.0, min_silence_len=1.0, keep_silence=0.1)\n\n            samples = samples.astype(np.float32)\n            batch_audio.append(samples)\n            batch_labels.append(np.array([label], dtype=np.float32))\n            # self.current_filenames.append(Path(audio_path).name)\n            batch_idx.append(np.array([self.i], dtype=np.int32))\n            self.i += 1\n        return batch_audio, batch_labels, batch_idx\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-29T15:05:56.116055Z","iopub.execute_input":"2025-05-29T15:05:56.116333Z","iopub.status.idle":"2025-05-29T15:05:56.138204Z","shell.execute_reply.started":"2025-05-29T15:05:56.116311Z","shell.execute_reply":"2025-05-29T15:05:56.137386Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nclass AudioDaliProcessor:\n    def __init__(self, df, audio_dir, cfg, device_id=0, max_duration=10, output_dir=None, reduce_human_voice=True, reduce_silence=True):\n        self.max_samples = max_duration * cfg.SAMPLE_RATE\n        self.audio_dir = Path(audio_dir)\n        self.cfg = cfg\n        self.df = df\n        self.reduce_human_voice = reduce_human_voice\n        self.reduce_silence = reduce_silence\n        self.batch_size = cfg.BATCH_SIZE\n        self.num_threads = cfg.NUM_THREAD\n        self.num_worker = cfg.NUM_WORKERS\n        self.device_id = device_id\n        self.output_dir = Path(output_dir) if output_dir else None\n\n        self.pipe = self._build_pipeline()\n        self.pipe.build()\n\n        self.iterator = DALIGenericIterator(\n            pipelines=self.pipe,\n            output_map=[\"mel\",\"label\", \"id\"],\n            auto_reset=True\n        )\n\n    @pipeline_def\n    def _pipeline(self):\n        audio, label, id = fn.external_source(\n            source = ExternalInputIterator(\n                df = self.df,\n                cfg = self.cfg,\n                reduce_human_voice=self.reduce_human_voice,\n                reduce_silence=self.reduce_silence\n            ),\n            num_outputs=3,\n            dtype=[types.FLOAT, types.FLOAT, types.INT32],\n            batch = True,\n            # parallel =True,\n        )\n        audio = audio.gpu()\n\n        if self.batch_size > 1:\n            audio = fn.slice(\n                audio,\n                start=0,\n                shape=self.max_samples,\n                axes=[0],\n                out_of_bounds_policy=\"pad\",\n                fill_values=0.0\n            )\n\n        audio = fn.normalize(audio, device=\"gpu\")\n\n        spectrogram = fn.spectrogram(\n            audio,\n            device=\"gpu\",\n            nfft=self.cfg.WINDOW_LENGTH,\n            window_length=self.cfg.WINDOW_LENGTH,\n            window_step=self.cfg.WINDOW_STEP,\n        )\n\n        mel = fn.mel_filter_bank(\n            spectrogram,\n            device=\"gpu\",\n            sample_rate=self.cfg.SAMPLE_RATE,\n            nfilter=self.cfg.NFILTER_MEL,\n            freq_low=self.cfg.FREQ_LOW,\n            freq_high=self.cfg.FREQ_HIGH,\n        )\n\n        mel_db = fn.to_decibels(\n            mel,\n            device=\"gpu\",\n            multiplier=10.0,\n            cutoff_db=-80\n        )\n\n        return mel_db, label, id\n\n    def _build_pipeline(self):\n        return self._pipeline(\n            batch_size=self.batch_size,\n            num_threads=self.num_threads,\n            device_id=self.device_id,\n            # py_num_workers=self.num_worker,\n            # py_start_method=\"spawn\",\n        )\n\n    def run(self, show_first=False):\n        all_mels = []\n        if self.output_dir:\n            self.output_dir.mkdir(parents=True, exist_ok=True)\n\n        print(\"🔄 Processing batches:\")\n        for batch in self.iterator:\n            mels = batch[0][\"mel\"].to(\"cpu\").numpy()\n            labels = batch[0][\"label\"].to(\"cpu\").numpy()\n            ids = batch[0][\"id\"].to(\"cpu\").numpy()\n            for mel, label, id in zip(mels, labels, ids):\n                row = self.df.iloc[id[0]]\n                fname = Path(row['ogg_path']).name\n                label_name = self.cfg.CLASS_NAME[int(label[0])]  # lấy label chính xác từ df\n                mel = mel.astype(np.float16)\n                all_mels.append(mel)\n\n                if self.output_dir:\n                    out_path = self.output_dir / label_name / (Path(fname).stem + \".npz\")\n                    out_path.parent.mkdir(parents=True, exist_ok=True)\n                    np.savez_compressed(out_path, mel=mel)\n\n                if show_first:\n                    self._plot_mel(mel)\n                    show_first = False\n\n        return all_mels\n\n    def _plot_mel(self, mel):\n        plt.figure(figsize=(10, 4))\n        plt.imshow(mel, aspect='auto', origin='lower', cmap='magma')\n        plt.title(\"Mel Spectrogram\")\n        plt.xlabel(\"Time Frames\")\n        plt.ylabel(\"Mel Bands\")\n        plt.colorbar(label=\"dB\")\n        plt.tight_layout()\n        plt.show()\n\nif __name__ == \"__main__\":\n    # Load metadata\n    start_ = datetime.now()\n    df = load_train_df(cfg, rating_threshold=0.0)\n    if df is None:\n        print(\"No metadata found.\")\n        exit(1)\n\n    # Load human voice data\n    human_voice_dict = load_human_voice(cfg.HUMAN_VOICE_PKL)\n    # Set up configuration for DALI processing\n    cfg.WINDOW_LENGTH = 2048\n    cfg.WINDOW_STEP = 1024\n    cfg.NFILTER_MEL = 256\n    cfg.FREQ_LOW = 150\n    cfg.FREQ_HIGH = 14000\n    cfg.BATCH_SIZE = 1\n    cfg.NUM_THREAD = 6\n    # cfg.NUM_WORKERS = 4 # only for training or loading\n\n    processor = AudioDaliProcessor(\n        df,\n        cfg=cfg,\n        audio_dir=cfg.DATA_PATH,\n        output_dir=cfg.OUTPUT_FOLDER,\n        reduce_human_voice=False,\n        reduce_silence=False,\n    )\n\n    mel_list = processor.run(show_first=True)\n    #save thời gian chạy vào file log.txt\n    with open(\"log.txt\", \"a\") as f:\n        f.write(f\"Start time: {start_}\\n\")\n        f.write(f\"End time: {datetime.now()}\\n\")\n        f.write(f\"Total spectrograms processed: {len(mel_list)}\\n\")\n    print(f\"✅ Total spectrograms processed: {len(mel_list)}\")\n    # os.system(\"sudo shutdown now\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-29T15:07:57.012742Z","iopub.execute_input":"2025-05-29T15:07:57.013277Z","execution_failed":"2025-05-29T15:12:08.112Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}