{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11421803,"sourceType":"datasetVersion","datasetId":7153191}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!nvidia-smi","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T12:58:54.584080Z","iopub.execute_input":"2025-04-17T12:58:54.584271Z","iopub.status.idle":"2025-04-17T12:58:54.805388Z","shell.execute_reply.started":"2025-04-17T12:58:54.584254Z","shell.execute_reply":"2025-04-17T12:58:54.804732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install noisereduce pedalboard\n!pip download --extra-index-url https://developer.download.nvidia.com/compute/redist/nightly nvidia-dali-nightly-cuda120\n!ls /kaggle/working |grep nvidia_dali_nightly_cuda120 |xargs pip install \nimport numpy as np\nimport matplotlib.pyplot as plt\n\n%matplotlib inline\nimport os, gc, random \nimport pandas as pd\nimport pickle\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\nimport IPython.display as ipd\nfrom IPython.display import display, clear_output\nimport ipywidgets as widgets\nimport librosa\nimport librosa.display\nimport soundfile as sf\nimport noisereduce as nr\nfrom pedalboard import Pedalboard, Gain\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, accuracy_score, confusion_matrix\nfrom nvidia.dali import pipeline_def\nimport nvidia.dali.fn as fn\nimport nvidia.dali.types as types\nimport nvidia.dali as dali\n\n!rm -rf /kaggle/working/\nclear_output()\nprint(\"Install and Import DONE\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-17T12:58:54.806622Z","iopub.execute_input":"2025-04-17T12:58:54.806890Z","iopub.status.idle":"2025-04-17T12:59:44.879091Z","shell.execute_reply.started":"2025-04-17T12:58:54.806864Z","shell.execute_reply":"2025-04-17T12:59:44.878242Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"from pathlib import Path\nclass Config:\n    def __init__(self, **kwargs):\n        for k, v in kwargs.items():\n            setattr(self, k, v)\n\n    def update(self, **kwargs):\n        for k, v in kwargs.items():\n            setattr(self, k, v)\n\n# Initialize and set basic configuration\ncfg = Config(\n    DATA_FOLD_IDX = 0, # 0,1,2,3 \n    SEED=42, \n    SAMPLE_RATE=32000,\n    DATA_PATH=Path(\"/kaggle/input/birdclef-2025/train_audio\"),\n    OUTPUT_FOLDER =Path(\"/kaggle/working/\"),\n    COLOR_MAP =['inferno'],\n    OUTPUT_METADATA=\"spec_img_meta.csv\",\n    WINDOW=\"hann\",\n    NFILTER_MEL=128,\n    WINDOW_LENGTH= 1024,\n    WINDOW_STEP= 512,\n    FREQ_HIGH=14000,\n    TARGET_DURATION_S = 5,\n    TARGET_SAMPLES = 5*32000,\n    DEVICE = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n    )\n# Function to seed everything to ensure reproducibility\ndef seed_everything(seed):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False # Change to true if input sizes are kept constant\n\nseed_everything(cfg.SEED)\n# Verifying changes\nprint(cfg.__dict__)\n# Device check\nprint(f\"Using device: {cfg.DEVICE}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T13:16:10.211013Z","iopub.execute_input":"2025-04-17T13:16:10.211544Z","iopub.status.idle":"2025-04-17T13:16:10.220294Z","shell.execute_reply.started":"2025-04-17T13:16:10.211523Z","shell.execute_reply":"2025-04-17T13:16:10.219605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Taking a closer look at the meta data\nmetadata_path = Path(\"/kaggle/input/birdclef-2025/train.csv\")\n\ndef split_df_into_parts(df, n_parts=4, seed=42):\n    # 1) shuffle reproducibly\n    df_shuffled = df.sample(frac=1, random_state=seed).reset_index(drop=True)\n    parts = np.array_split(df_shuffled, n_parts)\n    return parts\n    \nif metadata_path.exists():\n    train_df = pd.read_csv(metadata_path)\n    train_df = train_df.drop(columns=['url', 'license', 'common_name','collection','author','type'])\n    print(train_df.head(15))\n    print(\"\\nMetadata Columns:\", train_df.columns)\n    print(\"\\nTraining Samples:\", len(train_df))\n    print(\"\\nUnique Species:\", train_df['primary_label'].nunique())\n    print(\"\\nSecondary Species Labels(Recordist Marked):\", train_df['secondary_labels'].nunique())\n    # Key distributions\n    print(\"\\nSpecies distribution (top 10):\")\n    print(train_df['primary_label'].value_counts().head(10))\nelse:\n    print(f\"Metadata file not found at {meta_datapath}. Check path!\")\nlist_part = split_df_into_parts(train_df, n_parts=4, seed=42)\ntrain_df = list_part[cfg.DATA_FOLD_IDX]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T13:16:11.945247Z","iopub.execute_input":"2025-04-17T13:16:11.945814Z","iopub.status.idle":"2025-04-17T13:16:12.067550Z","shell.execute_reply.started":"2025-04-17T13:16:11.945789Z","shell.execute_reply":"2025-04-17T13:16:12.066785Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T13:16:37.005927Z","iopub.execute_input":"2025-04-17T13:16:37.006625Z","iopub.status.idle":"2025-04-17T13:16:37.017782Z","shell.execute_reply.started":"2025-04-17T13:16:37.006600Z","shell.execute_reply":"2025-04-17T13:16:37.017006Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.describe(include=[np.number])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T13:16:14.642146Z","iopub.execute_input":"2025-04-17T13:16:14.642704Z","iopub.status.idle":"2025-04-17T13:16:14.668897Z","shell.execute_reply.started":"2025-04-17T13:16:14.642681Z","shell.execute_reply":"2025-04-17T13:16:14.668332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scientific_name_counts = train_df['scientific_name'].value_counts()\nprint(scientific_name_counts)\noverlap_map = {name: 0.8 if count < 50 else 0.2 for name, count in scientific_name_counts.items()}\nprint(\"dataframe mới\")\ntrain_df['overlap-percent'] = train_df['scientific_name'].map(overlap_map)\nprint(train_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T03:56:38.131355Z","iopub.execute_input":"2025-04-17T03:56:38.131567Z","iopub.status.idle":"2025-04-17T03:56:38.144596Z","shell.execute_reply.started":"2025-04-17T03:56:38.131549Z","shell.execute_reply":"2025-04-17T03:56:38.143923Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Analyzing audio durations...\")\ndurations = []\npbar = tqdm(train_df['filename'].tolist(), desc=\"Calculating durations\")\nfor filename in pbar:\n    file_path = cfg.DATA_PATH/filename\n    if file_path.exists():\n        try:\n            # Efficient approach to get duration with loading the whole file\n            info = sf.info(file_path)\n            durations.append(info.duration)\n        except Exception as e:\n            print(f\"Could not get info for {filename}: {e}\") #Comment / uncomment for debugging\n            durations.append(np.nan) # mark errors\n    else:\n        durations.append(np.nan)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T03:56:38.146567Z","iopub.execute_input":"2025-04-17T03:56:38.146791Z","iopub.status.idle":"2025-04-17T04:03:49.153714Z","shell.execute_reply.started":"2025-04-17T03:56:38.146773Z","shell.execute_reply":"2025-04-17T04:03:49.152754Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df[\"duration\"]= durations\ntrain_df.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T04:03:49.154713Z","iopub.execute_input":"2025-04-17T04:03:49.155329Z","iopub.status.idle":"2025-04-17T04:03:49.171236Z","shell.execute_reply.started":"2025-04-17T04:03:49.155293Z","shell.execute_reply":"2025-04-17T04:03:49.170593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"duration_by_class = train_df.groupby('primary_label')['duration'].sum()\nduration_by_class = duration_by_class.reset_index().sort_values(by=['duration']).reset_index()\n# Hiển thị dữ liệu tổng hợp (có thể in ra để kiểm tra)\nprint(duration_by_class)\n\n# Vẽ biểu đồ cột:\nplt.figure(figsize=(10, 6))\nplt.bar(duration_by_class['primary_label'].astype(str), duration_by_class['duration'], color='skyblue')\nplt.xlabel('Primary Label')\nplt.ylabel('total time (second)')\nplt.title('Total Time Per Class (primary_label)')\nplt.xticks(rotation=45)\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T04:05:14.88518Z","iopub.execute_input":"2025-04-17T04:05:14.885942Z","iopub.status.idle":"2025-04-17T04:05:16.537435Z","shell.execute_reply.started":"2025-04-17T04:05:14.885916Z","shell.execute_reply":"2025-04-17T04:05:16.53651Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ============================\n# 1. Hàm load và tiền xử lý audio\n# ============================\ndef load_and_preprocess(audio_path, sr):\n    \"\"\"\n    - Load file .ogg bằng librosa.\n    - Giảm nhiễu bằng noisereduce.\n    - Tăng âm lượng bằng pedalboard.\n    \"\"\"\n    samples, sr = librosa.load(audio_path, sr=sr)\n    samples_nr = nr.reduce_noise(y=samples, sr=sr)\n    board = Pedalboard([Gain(gain_db=10)])\n    samples_proc = board(samples_nr, sr)\n    return samples_proc, sr\n\n# ============================\n# 2. Hàm tạo sliding window (5 giây, bước nhảy 0.5 giây)\n# ============================\ndef get_sliding_windows(audio, sr, segment_duration=5.0, overlap_percent=0.5):\n    \"\"\"\n    Trả về danh sách các tuple (start_sample, end_sample) cho mỗi window.\n    \"\"\"\n    step = segment_duration*(1-overlap_percent)\n    seg_samples = int(segment_duration * sr)\n    step_samples = int(step * sr)\n    windows = []\n    for start in range(0, len(audio) - seg_samples + 1, step_samples):\n        end = start + seg_samples\n        windows.append((start, end))\n    return windows\n    \n# ============================\n# 3. Hàm loại bỏ đoạn có tiếng người và padding\n# ============================\ndef remove_human_voice(segment_audio, sr, window_start, human_voice_intervals, seg_duration=5.0):\n    \"\"\"\n    - segment_audio: mảng audio của 1 window (5 giây)\n    - window_start: thời gian bắt đầu (giây) của window trong file gốc.\n    - human_voice_intervals: danh sách dict {'start': ..., 'end': ...} (giây) đã phát hiện tiếng người trong file gốc.\n    \n    Hàm này sẽ loại bỏ (cut) các đoạn trùng với tiếng người, nối các đoạn còn lại lại,\n    rồi thêm padding (âm im) ở 2 đầu để độ dài về 5 giây.\n    \"\"\"\n    seg_len = len(segment_audio)\n    \n    # Lọc các interval có giao với window hiện tại\n    intervals = [iv for iv in human_voice_intervals \n                 if iv['start'] < window_start + seg_duration and iv['end'] > window_start]\n    intervals = sorted(intervals, key=lambda x: x['start'])\n    \n    # Nếu không có tiếng người trong window -> return segment ban đầu\n    if not intervals:\n        return segment_audio\n    \n    nonhuman_parts = []\n    current_sample = 0\n    # Xử lý các interval theo thứ tự tăng dần\n    for iv in intervals:\n        # Chuyển đổi thời gian của interval về thời gian tương đối trong window\n        start_relative = max(iv['start'] - window_start, 0)\n        end_relative = min(iv['end'] - window_start, seg_duration)\n        start_idx = int(start_relative * sr)\n        end_idx = int(end_relative * sr)\n        # Nếu có đoạn không có tiếng người trước interval hiện tại, lấy nó\n        if start_idx > current_sample:\n            nonhuman_parts.append(segment_audio[current_sample:start_idx])\n        # Cập nhật vị trí hiện tại (bỏ qua phần có tiếng người)\n        current_sample = end_idx\n    # Lấy phần sau interval cuối cùng\n    if current_sample < seg_len:\n        nonhuman_parts.append(np.zeros(seg_len -current_sample))\n        nonhuman_parts.append(segment_audio[current_sample:])\n    \n    if nonhuman_parts:\n        nonhuman_audio = np.concatenate(nonhuman_parts)\n    else:\n        nonhuman_audio = np.array([], dtype=segment_audio.dtype)\n    \n    # Padding thêm silence để có đủ số mẫu ban đầu (5 giây)\n    desired_length = seg_len\n    current_length = len(nonhuman_audio)\n    if current_length < desired_length:\n        pad_total = desired_length - current_length\n        pad_left = pad_total // 2\n        pad_right = pad_total - pad_left\n        nonhuman_audio = np.concatenate([\n            np.zeros(pad_left, dtype=segment_audio.dtype), \n            nonhuman_audio, \n            np.zeros(pad_right, dtype=segment_audio.dtype)\n        ])\n    else:\n        nonhuman_audio = nonhuman_audio[:desired_length]\n    \n    return nonhuman_audio\n\n# ============================\n# 4. Lưu ảnh Mel Spectrogram\n# ============================\nfrom PIL import Image\nimport matplotlib.cm as cm\n\n\ndef save_spectrogram_image(mel_spec, cfg, output_dir, output_filename, output_size=(1280, 720)):\n    # Normalize mel spectrogram to 0–1\n    mel_spec = (mel_spec - mel_spec.min()) / (mel_spec.max() - mel_spec.min())\n    # List of colormaps\n    colormaps = cfg.COLOR_MAP\n\n    for cmap_name in colormaps:\n        out_path =  cfg.OUTPUT_FOLDER /cmap_name /output_dir\n        out_path.mkdir(parents=True, exist_ok=True)\n        output_file = out_path / output_filename\n        \n        cmap = cm.get_cmap(cmap_name)\n        rgba_img = cmap(mel_spec)  # shape: (H, W, 4), values in [0, 1]\n        rgb_img = (rgba_img[:, :, :3] * 255).astype(np.uint8)\n        img = Image.fromarray(rgb_img)\n\n        # Flip vertically (to match origin='lower' behavior)\n        img = img.transpose(Image.FLIP_TOP_BOTTOM)\n        img_resized = img.resize(output_size, Image.LANCZOS)\n\n        # Save final image\n        img_resized.save(output_file)\n\n\ndef show_spectrogram(spec, title, sr, hop_length, y_axis=\"log\", x_axis=\"time\"):\n    librosa.display.specshow(\n        spec, sr=sr, y_axis=y_axis, x_axis=x_axis, hop_length=hop_length\n    )\n    plt.title(title)\n    plt.colorbar(format=\"%+2.0f dB\")\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T04:19:01.174511Z","iopub.execute_input":"2025-04-17T04:19:01.174809Z","iopub.status.idle":"2025-04-17T04:19:01.188797Z","shell.execute_reply.started":"2025-04-17T04:19:01.174787Z","shell.execute_reply":"2025-04-17T04:19:01.187915Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#======================================================================================\n# ********************************Test output 1 file***********************************\n#=====================================================================================\n# y, sr = librosa.load(\"/kaggle/input/birdclef-2025/train_audio/1139490/CSA36385.ogg\")\n# y = y[:32000*5]\ny,sr = load_and_preprocess(\"/kaggle/input/birdclef-2025/train_audio/1139490/CSA36385.ogg\", 32000)\ny = y[32000*0:32000*5]\naudio_data = np.array(y, dtype=np.float32)\n\n@pipeline_def\ndef mel_spectrogram_pipe(nfft, window_length, window_step,sample_rate,nfilter,freq_high,device=\"gpu\"):\n    audio = types.Constant(device=device, value=audio_data)\n    spectrogram = fn.spectrogram(\n        audio,\n        device=device,\n        nfft=nfft,\n        window_length=window_length,\n        window_step=window_step,\n    )\n    mel_spectrogram = fn.mel_filter_bank(\n        spectrogram, sample_rate=sr, nfilter=nfilter, freq_high=freq_high\n    )\n    mel_spectrogram_dB = fn.to_decibels(\n        mel_spectrogram, multiplier=10.0, cutoff_db=-80\n    )\n    return mel_spectrogram_dB\n    \npipe = mel_spectrogram_pipe(\n    device=\"gpu\",\n    batch_size=1,\n    num_threads=3,\n    device_id=0,\n    nfft=cfg.WINDOW_LENGTH,\n    window_length=cfg.WINDOW_LENGTH,\n    window_step=cfg.WINDOW_STEP,\n    sample_rate=cfg.SAMPLE_RATE,\n    nfilter=cfg.NFILTER_MEL,\n    freq_high=cfg.FREQ_HIGH,\n)\npipe.build()\noutputs = pipe.run()\nmel_spectrogram_dali_db = np.array(outputs[0][0].as_cpu())\n\nplt.imshow(mel_spectrogram_dali_db)\n\nprint(mel_spectrogram_dali_db.shape)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T04:25:41.25963Z","iopub.execute_input":"2025-04-17T04:25:41.260106Z","iopub.status.idle":"2025-04-17T04:25:43.573602Z","shell.execute_reply.started":"2025-04-17T04:25:41.260078Z","shell.execute_reply":"2025-04-17T04:25:43.572982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# -------------------------\n# 4. DALI Pipeline: Tạo Mel Spectrogram\n# -------------------------\n\ndef compute_mel_spectrogram(audio_segment, cfg):\n    audio_data = np.array(audio_segment, dtype=np.float32)\n    @pipeline_def\n    def mel_spectrogram_pipe(nfft, window_length, window_step,sample_rate,nfilter,freq_high,device=\"gpu\"):\n        audio = types.Constant(device=device, value=audio_data)\n        spectrogram = fn.spectrogram(\n            audio,\n            device=device,\n            nfft=nfft,\n            window_length=window_length,\n            window_step=window_step,\n        )\n        mel_spectrogram = fn.mel_filter_bank(\n            spectrogram, sample_rate=sr, nfilter=nfilter, freq_high=freq_high\n        )\n        mel_spectrogram_dB = fn.to_decibels(\n            mel_spectrogram, multiplier=10.0, cutoff_db=-80\n        )\n        return mel_spectrogram_dB\n        \n    pipe = mel_spectrogram_pipe(\n        device=\"gpu\",\n        batch_size=1,\n        num_threads=3,\n        device_id=0,\n        nfft=cfg.WINDOW_LENGTH,\n        window_length=cfg.WINDOW_LENGTH,\n        window_step=cfg.WINDOW_STEP,\n        sample_rate=cfg.SAMPLE_RATE,\n        nfilter=cfg.NFILTER_MEL,\n        freq_high=cfg.FREQ_HIGH,\n    )\n    pipe.build()\n    outputs = pipe.run()\n    mel_spectrogram_dali_db = np.array(outputs[0][0].as_cpu())\n\n    return mel_spectrogram_dali_db\n\n\n\ndef process_audio_file(audio_path, human_voice_dict, output_folder, cfg, overlap_percent):\n    samples_proc, sr = load_and_preprocess(audio_path, cfg.SAMPLE_RATE)\n    # Tạo sliding windows, mỗi window TARGET_DURATION_S giây, bước 0.5 giây\n    windows = get_sliding_windows(samples_proc, sr, cfg.TARGET_DURATION_S, overlap_percent = overlap_percent)\n    \n    # Lấy human voice intervals nếu tồn tại cho file này (dùng đường dẫn tương đối so với DATA_PATH)\n    human_intervals = human_voice_dict.get(audio_path, [])\n    for start_sample, end_sample in windows:\n        window_start_time = start_sample / sr\n        segment_audio = samples_proc[start_sample:end_sample]\n        # Loại bỏ tiếng người và padding lại thành đủ TARGET_DURATION_S giây\n        segment_clean = remove_human_voice(segment_audio, sr, window_start_time, human_intervals, cfg.TARGET_DURATION_S)\n        # Tính Mel Spectrogram sử dụng NVIDIA DALI\n        mel_spec = compute_mel_spectrogram(segment_clean, cfg)\n        # Tạo tên file: AudioFileName_start_end.jpg\n        base_name = os.path.splitext(Path(audio_path).name)[0]\n        output_filename = f\"{base_name}_{window_start_time:.1f}_{(window_start_time+cfg.TARGET_DURATION_S):.1f}.jpg\"\n        #Save into virisdis format\n        save_spectrogram_image(mel_spec, cfg, output_folder, output_filename, output_size=(640, 480))\n        # save_spectrogram_image(mel_spec, str(output_file), cmap='gray')\n        print(f\"Saved {output_filename}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T04:21:34.678272Z","iopub.execute_input":"2025-04-17T04:21:34.678819Z","iopub.status.idle":"2025-04-17T04:21:34.686494Z","shell.execute_reply.started":"2025-04-17T04:21:34.678797Z","shell.execute_reply":"2025-04-17T04:21:34.685897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\nwith open(\"/kaggle/input/bc25-human-detect-sound/train_voice_data.pkl\", \"rb\") as f:\n    human_voice_dict = pickle.load(f)\n        \n    # Lặp qua từng dòng trong DataFrame và xử lý file audio\nos.makedirs(cfg.OUTPUT_FOLDER, exist_ok=True)\nerror_list = []\nfor cmap_name in cfg.COLOR_MAP:\n    path =  cfg.OUTPUT_FOLDER /cmap_name \n    path.mkdir(parents=True, exist_ok=True)\nfor idx, row in train_df.iterrows():\n    output_folder =  row['primary_label']\n    os.makedirs(output_folder, exist_ok=True)\n    rel_path = row['filename']  # đường dẫn tương đối\n    overlap_percent = row['overlap-percent']\n    audio_path = cfg.DATA_PATH / rel_path\n    print(f\"Processing {audio_path} ...\")\n    try:\n        process_audio_file(str(audio_path), human_voice_dict, output_folder, cfg, overlap_percent)\n    except Exception as e:\n        print_out_log = f\"Error processing {audio_path}: {e}\"\n        error_list.append(print_out_log)\n        print(f\"Error processing {audio_path}: {e}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T04:21:36.091856Z","iopub.execute_input":"2025-04-17T04:21:36.092563Z","iopub.status.idle":"2025-04-17T04:21:53.936315Z","shell.execute_reply.started":"2025-04-17T04:21:36.092539Z","shell.execute_reply":"2025-04-17T04:21:53.935539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\ndisplay(Image.open(\"/kaggle/working/plasma/1139490/CSA36385_0._5.0.jpg\"))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-17T04:24:03.103839Z","iopub.execute_input":"2025-04-17T04:24:03.104447Z","iopub.status.idle":"2025-04-17T04:24:03.158614Z","shell.execute_reply.started":"2025-04-17T04:24:03.10442Z","shell.execute_reply":"2025-04-17T04:24:03.157907Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(error_list)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}