{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **BirdCLEF 2025 Data Preprocessing Notebook**\nThis notebook demonstrates how we can transform audio data into mel-spectrogram data. This transformation is essential for training 2D Convolutional Neural Networks (CNNs) on audio data, as it converts the one-dimensional audio signals into two-dimensional image-like representations.\nI run this public notebook in debug mode(only a few sample processing). You can find the fully preprocessed mel spectrogram training dataset here --> [BirdCLEF'25 | Mel Spectrograms](https://www.kaggle.com/datasets/kadircandrisolu/birdclef25-mel-spectrograms).\n","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport warnings\nimport logging\nimport time\nimport math\nimport cv2\nimport random \nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport librosa\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\n\nwarnings.filterwarnings(\"ignore\")\nlogging.basicConfig(level=logging.ERROR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T06:55:29.745076Z","iopub.execute_input":"2025-05-06T06:55:29.745736Z","iopub.status.idle":"2025-05-06T06:55:29.754703Z","shell.execute_reply.started":"2025-05-06T06:55:29.745704Z","shell.execute_reply":"2025-05-06T06:55:29.753981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Configuration ---\nclass Config:\n    DEBUG_MODE = False\n    N_MAX_DEBUG = 100 \n\n    # --- OUTPUT ---\n    OUTPUT_DIR = '/kaggle/working/'\n    OUTPUT_NPY_FILE = os.path.join(OUTPUT_DIR, f'birdclef25_melspec_5s_randcrop_32k_2048fft_512hop_128mel_rs256.npy')\n\n    # --- INPUT DATA ---\n    DATA_ROOT = '/kaggle/input/birdclef-2025'\n    AUDIO_SUBDIR = 'train_audio' \n\n    # --- AUDIO PARAMETERS ---\n    FS = 32000\n    TARGET_DURATION = 5.0\n    USE_RANDOM_CROP = True \n\n    # --- MEL SPECTROGRAM PARAMETERS ---\n    N_FFT = 2048\n    HOP_LENGTH = 512\n    WIN_LENGTH = 2048 \n    N_MELS = 128  \n    FMIN = 20\n    FMAX = 16000\n\n    # --- TARGET SHAPE & RESIZE ---\n    DO_RESIZE = True #\n    TARGET_SHAPE = (256, 256)\n\nconfig = Config()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T06:56:06.352573Z","iopub.execute_input":"2025-05-06T06:56:06.352853Z","iopub.status.idle":"2025-05-06T06:56:06.358417Z","shell.execute_reply.started":"2025-05-06T06:56:06.352831Z","shell.execute_reply":"2025-05-06T06:56:06.357770Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Setup ---\nprint(f\"--- Configuration ---\")\nprint(f\"DEBUG_MODE: {'ON' if config.DEBUG_MODE else 'OFF'} ({config.N_MAX_DEBUG if config.DEBUG_MODE else 'ALL'} samples)\")\nprint(f\"Target Duration: {config.TARGET_DURATION}s\")\nprint(f\"Use Random Crop: {config.USE_RANDOM_CROP}\")\nprint(f\"Sample Rate: {config.FS}\")\nprint(f\"Mel Params: N_FFT={config.N_FFT}, HOP={config.HOP_LENGTH}, WIN={config.WIN_LENGTH}, N_MELS={config.N_MELS}, FMIN={config.FMIN}, FMAX={config.FMAX}\")\nprint(f\"Resize Spectrogram: {config.DO_RESIZE}\")\nif config.DO_RESIZE:\n    print(f\"Target Shape (after resize): {config.TARGET_SHAPE}\")\nelse:\n    calculated_shape = (config.N_MELS, int(np.floor((config.TARGET_DURATION * config.FS) / config.HOP_LENGTH) + 1))\n    print(f\"Target Shape (original): {calculated_shape}\")\n    config.TARGET_SHAPE = calculated_shape \nprint(f\"Output NPY file: {config.OUTPUT_NPY_FILE}\")\nprint(f\"--------------------\")\n\nos.makedirs(config.OUTPUT_DIR, exist_ok=True)\n\nprint(\"Loading training metadata...\")\ntrain_df = pd.read_csv(f'{config.DATA_ROOT}/train.csv')\nworking_df = train_df[['filename']].copy() \nworking_df['filepath'] = config.DATA_ROOT + '/' + config.AUDIO_SUBDIR + '/' + working_df.filename\n# *** Tạo samplename khớp với training: bỏ '.ogg' ***\nworking_df['samplename'] = working_df.filename.map(lambda x: x.replace('.ogg',''))\n\ntotal_samples_available = len(working_df)\nsamples_to_process = min(total_samples_available, config.N_MAX_DEBUG if config.DEBUG_MODE else total_samples_available)\n\nprint(f'Total samples to process: {samples_to_process} out of {total_samples_available} available')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T06:56:08.771876Z","iopub.execute_input":"2025-05-06T06:56:08.772169Z","iopub.status.idle":"2025-05-06T06:56:08.969882Z","shell.execute_reply.started":"2025-05-06T06:56:08.772147Z","shell.execute_reply":"2025-05-06T06:56:08.969263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Hàm xử lý Audio ---\ndef audio_to_melspectrogram(audio_data, config):\n    \"\"\"Converts audio data to a Mel spectrogram according to Config.\"\"\"\n    if np.isnan(audio_data).any():\n        # print(f\"Warning: NaN found in audio data, replacing with mean.\")\n        mean_signal = np.nanmean(audio_data)\n        audio_data = np.nan_to_num(audio_data, nan=mean_signal if not np.isnan(mean_signal) else 0.0)\n\n    mel_spec = librosa.feature.melspectrogram(\n        y=audio_data,\n        sr=config.FS,\n        n_fft=config.N_FFT,\n        hop_length=config.HOP_LENGTH,\n        win_length=config.WIN_LENGTH,\n        n_mels=config.N_MELS,\n        fmin=config.FMIN,\n        fmax=config.FMAX,\n        power=2.0,\n        center=True,\n        pad_mode=\"reflect\"\n    )\n\n    # Convert to decibels and normalize to [0, 1]\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n    if mel_spec_db.max() == mel_spec_db.min(): # Handle silent clips\n        return np.zeros((config.N_MELS, int(np.floor((config.TARGET_DURATION * config.FS) / config.HOP_LENGTH) + 1)), dtype=np.float32)\n    mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n\n    return mel_spec_norm\n\ndef process_single_file(filepath, samplename, config):\n    \"\"\"Loads audio, extracts/pads 5s, creates Mel spectrogram, and resizes if needed.\"\"\"\n    try:\n        audio_data, _ = librosa.load(filepath, sr=config.FS)\n\n        target_samples = int(config.TARGET_DURATION * config.FS)\n        current_samples = len(audio_data)\n        processed_audio = np.zeros(target_samples, dtype=np.float32) # Khởi tạo mảng 0\n\n        if current_samples == 0:\n            # print(f\"Warning: Empty audio file {filepath}. Returning zero spec.\")\n            # Tạo spec zero với shape gốc trước resize\n            orig_shape = (config.N_MELS, int(np.floor(target_samples / config.HOP_LENGTH) + 1))\n            mel_spec_norm = np.zeros(orig_shape, dtype=np.float32)\n\n        elif current_samples < target_samples:\n            # Pad audio ngắn\n            start = random.randint(0, target_samples - current_samples) # Pad ngẫu nhiên 2 đầu\n            processed_audio[start : start + current_samples] = audio_data\n            mel_spec_norm = audio_to_melspectrogram(processed_audio, config)\n\n        elif current_samples == target_samples:\n            processed_audio = audio_data\n            mel_spec_norm = audio_to_melspectrogram(processed_audio, config)\n\n        else: \n            if config.USE_RANDOM_CROP:\n                # *** Lấy 5 giây ngẫu nhiên ***\n                max_start_idx = current_samples - target_samples\n                start_idx = random.randint(0, max_start_idx)\n                processed_audio = audio_data[start_idx : start_idx + target_samples]\n            else:\n                processed_audio = audio_data[:target_samples]\n            mel_spec_norm = audio_to_melspectrogram(processed_audio, config)\n\n        # --- Resize ---\n        final_spec = mel_spec_norm\n        if config.DO_RESIZE:\n            if final_spec.shape[1] == 0:\n                 print(f\"Warning: Spectrogram for {samplename} has zero width before resize. Creating zero array.\")\n                 final_spec = np.zeros(config.TARGET_SHAPE, dtype=np.float32)\n            elif final_spec.shape != config.TARGET_SHAPE:\n                 # cv2.resize cần (width, height)\n                 final_spec = cv2.resize(final_spec, (config.TARGET_SHAPE[1], config.TARGET_SHAPE[0]), interpolation=cv2.INTER_LINEAR)\n\n\n        return samplename, final_spec.astype(np.float32)\n\n    except Exception as e:\n        # print(f\"Error processing {filepath}: {e}\")\n        return samplename, None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T06:56:13.828648Z","iopub.execute_input":"2025-05-06T06:56:13.828915Z","iopub.status.idle":"2025-05-06T06:56:13.839787Z","shell.execute_reply.started":"2025-05-06T06:56:13.828895Z","shell.execute_reply":"2025-05-06T06:56:13.839071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Vòng lặp xử lý chính ---\nprint(\"\\nProcessing audio files...\")\nstart_time = time.time()\n\nall_spectrograms = {}\nerror_count = 0\n\nprocess_df = working_df.head(samples_to_process)\n\nfor _, row in tqdm(process_df.iterrows(), total=samples_to_process):\n    samplename, spec = process_single_file(row['filepath'], row['samplename'], config)\n    if spec is not None:\n        all_spectrograms[samplename] = spec\n    else:\n        error_count += 1\n\n# --- Kết thúc và Lưu ---\nend_time = time.time()\nprint(f\"\\nProcessing finished in {(end_time - start_time):.2f} seconds.\")\nprint(f\"Successfully generated {len(all_spectrograms)} spectrograms.\")\nif error_count > 0:\n    print(f\"Encountered errors in {error_count} files.\")\n\n# --- Lưu file NPY ---\nif all_spectrograms:\n    print(f\"\\nSaving spectrogram data to {config.OUTPUT_NPY_FILE}...\")\n    np.save(config.OUTPUT_NPY_FILE, all_spectrograms, allow_pickle=True)\n    print(f\"Data saved successfully. Shape of first spec: {next(iter(all_spectrograms.values())).shape}\")\nelse:\n    print(\"\\nNo spectrogram data generated to save.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T06:56:17.507908Z","iopub.execute_input":"2025-05-06T06:56:17.508652Z","iopub.status.idle":"2025-05-06T07:24:13.441226Z","shell.execute_reply.started":"2025-05-06T06:56:17.508625Z","shell.execute_reply":"2025-05-06T07:24:13.440334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Visualize ---\nif config.DEBUG_MODE and all_spectrograms: \n    print(\"\\nVisualizing examples...\")\n    visualize_keys = random.sample(list(all_spectrograms.keys()), min(4, len(all_spectrograms)))\n    plt.figure(figsize=(12, 5 * (len(visualize_keys)//2 + len(visualize_keys)%2))) \n    for i, key in enumerate(visualize_keys):\n        plt.subplot( (len(visualize_keys)+1)//2 , 2, i + 1)\n        spec_to_show = all_spectrograms[key]\n        if config.DO_RESIZE:\n             plt.imshow(spec_to_show, aspect='auto', origin='lower', cmap='viridis')\n             plt.title(f\"{key}\\nShape: {spec_to_show.shape} (Resized)\")\n             plt.colorbar()\n        else:\n             librosa.display.specshow(spec_to_show, sr=config.FS, hop_length=config.HOP_LENGTH,\n                                     x_axis='time', y_axis='mel', fmin=config.FMIN, fmax=config.FMAX, cmap='viridis')\n             plt.title(f\"{key}\\nShape: {spec_to_show.shape}\")\n             plt.colorbar(format='%+2.0f dB') \n\n    plt.tight_layout()\n    plt.savefig(os.path.join(config.OUTPUT_DIR, 'debug_melspec_examples.png'))\n    print(f\"Visualization saved to {os.path.join(config.OUTPUT_DIR, 'debug_melspec_examples.png')}\")\n    plt.show() ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T07:45:40.708134Z","iopub.execute_input":"2025-05-06T07:45:40.708410Z","iopub.status.idle":"2025-05-06T07:45:40.715201Z","shell.execute_reply.started":"2025-05-06T07:45:40.708392Z","shell.execute_reply":"2025-05-06T07:45:40.714384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\ndataset_name = \"melspec-train-audio-update\"\n\nos.makedirs(dataset_name, exist_ok=True)\n!cp birdclef25_melspec_5s_randcrop_32k_2048fft_512hop_128mel_rs256.npy {dataset_name}/","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T08:03:02.569172Z","iopub.execute_input":"2025-05-06T08:03:02.569816Z","iopub.status.idle":"2025-05-06T08:03:34.993038Z","shell.execute_reply.started":"2025-05-06T08:03:02.569788Z","shell.execute_reply":"2025-05-06T08:03:34.991956Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nos.environ['KAGGLE_USERNAME'] = ''\nos.environ['KAGGLE_KEY'] = ''","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T08:08:21.754832Z","iopub.execute_input":"2025-05-06T08:08:21.755581Z","iopub.status.idle":"2025-05-06T08:08:21.759861Z","shell.execute_reply.started":"2025-05-06T08:08:21.755551Z","shell.execute_reply":"2025-05-06T08:08:21.758966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"metadata = {\n    \"title\": \"MelSpec Train Audio\",\n    \"id\": f\"{os.environ['KAGGLE_USERNAME']}/{dataset_name}\",\n    \"licenses\": [{\"name\": \"CC0-1.0\"}]\n}\n\nimport json\nwith open(f\"{dataset_name}/dataset-metadata.json\", \"w\") as f:\n    json.dump(metadata, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T08:08:23.833764Z","iopub.execute_input":"2025-05-06T08:08:23.834515Z","iopub.status.idle":"2025-05-06T08:08:23.839384Z","shell.execute_reply.started":"2025-05-06T08:08:23.834490Z","shell.execute_reply":"2025-05-06T08:08:23.838572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!kaggle datasets create -p {dataset_name}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-06T08:08:26.152772Z","iopub.execute_input":"2025-05-06T08:08:26.153327Z","iopub.status.idle":"2025-05-06T08:12:26.077141Z","shell.execute_reply.started":"2025-05-06T08:08:26.153300Z","shell.execute_reply":"2025-05-06T08:12:26.076170Z"}},"outputs":[],"execution_count":null}]}