{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31040,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Transforming Audio-to-Mel Sectrogram**\n\nThe purpose of this Notebook is to preprocess and convert bird call data used in Kaggle's BirdCLEF 2025 competition into a format that can be used in machine learning models.\n\nBy running this Notebook, the birdsong data will be converted and stored in a mel-spectrogram format suitable as input to the machine learning model.","metadata":{}},{"cell_type":"markdown","source":"## Import Libraries","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2 # OpenCV library for image processing and computer vision\nimport math\nimport time\nimport librosa # Audio analysis library\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm # A library for displaying progress bars for loops\n\nimport torch # PyTorch machine learning framework\nimport warnings\n# warnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T23:13:36.891527Z","iopub.execute_input":"2025-05-16T23:13:36.891811Z","iopub.status.idle":"2025-05-16T23:13:42.837179Z","shell.execute_reply.started":"2025-05-16T23:13:36.891788Z","shell.execute_reply":"2025-05-16T23:13:42.836249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n    # If set to True, it limits the number of samples to be processed in subsequent data processing, speeding up development and testing.\n    DEBUG_MODE: bool = False\n\n    OUTPUT_DIR: str = \"/kaggle/working\"\n    DATA_ROOT: str = \"/kaggle/input/birdclef-2025\"\n\n    # Set the sampling rate of the speech data to 32 kHz.\n    FS: int = 32000\n\n    # Mel spectrogram parameters\n\n    # The Fast Fourier Transform (FFT) window size when performing the Sort Time Fourier Transform(STFT).\n    # The larget the window size, the higher the frequency resolution, but the lower the time resolution.\n    N_FFT: int = 1024\n\n    # The STFT window hop size (movement).\n    # The smaller the value, the higher the time resolution, but the higher the computational cost.\n    HOP_LENGTH: int = 512\n\n    # The number of mel filter banks in the mel spectrogram.\n    # It will be the height (number of frequency bins) of the generated mel spectrogram.\n    N_MELS: int = 128\n\n    # The lowest frequency used in the mel spectrogram calculation (unit: Hz).\n    FMIN: int = 50\n\n    # The highest frequency used in the mel spectrogram calculation (unit: Hz).\n    FMAX: int = 14000\n\n    # The target length in seconds for processing audio data.\n    # Based on this value, the audio data is trimmed.\n    TARGET_DURATION: float = 5.0\n\n    # The target shape (height, width) of the generated mel-spectrogram image.\n    # After audio processing, the image will be reized to this shape.\n    TARGET_SHAPE: tuple[int, int] = (256, 256)\n\n    # The maximum number of samples to be processed.\n    N_MAX: int | None = 50 if DEBUG_MODE else None\n\nconfig: Config = Config()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T23:13:42.838737Z","iopub.execute_input":"2025-05-16T23:13:42.839610Z","iopub.status.idle":"2025-05-16T23:13:42.846283Z","shell.execute_reply.started":"2025-05-16T23:13:42.839585Z","shell.execute_reply":"2025-05-16T23:13:42.845210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Debug mode: {config.DEBUG_MODE}\")\nprint(f\"Max samples to process: {config.N_MAX if config.N_MAX is not None else 'ALL'}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T23:13:42.847251Z","iopub.execute_input":"2025-05-16T23:13:42.847550Z","iopub.status.idle":"2025-05-16T23:13:42.868923Z","shell.execute_reply.started":"2025-05-16T23:13:42.847527Z","shell.execute_reply":"2025-05-16T23:13:42.868058Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Load Data","metadata":{}},{"cell_type":"code","source":"# Load taxonomy (information about biological classification systems) data\ntaxonomy_df: pd.DataFrame = pd.read_csv(f\"{config.DATA_ROOT}/taxonomy.csv\")\nspecies_class_map: dict[str, str] = dict(zip(taxonomy_df[\"primary_label\"], taxonomy_df[\"class_name\"]))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T23:14:30.853962Z","iopub.execute_input":"2025-05-16T23:14:30.854830Z","iopub.status.idle":"2025-05-16T23:14:30.878014Z","shell.execute_reply.started":"2025-05-16T23:14:30.854802Z","shell.execute_reply":"2025-05-16T23:14:30.877072Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"taxonomy_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T23:14:31.878105Z","iopub.execute_input":"2025-05-16T23:14:31.878426Z","iopub.status.idle":"2025-05-16T23:14:31.900709Z","shell.execute_reply.started":"2025-05-16T23:14:31.878403Z","shell.execute_reply":"2025-05-16T23:14:31.899890Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load training metadata\ntrain_df: pd.DataFrame = pd.read_csv(f\"{config.DATA_ROOT}/train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T23:15:37.646190Z","iopub.execute_input":"2025-05-16T23:15:37.646516Z","iopub.status.idle":"2025-05-16T23:15:37.819039Z","shell.execute_reply.started":"2025-05-16T23:15:37.646492Z","shell.execute_reply":"2025-05-16T23:15:37.818003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T23:15:42.777449Z","iopub.execute_input":"2025-05-16T23:15:42.777764Z","iopub.status.idle":"2025-05-16T23:15:42.792275Z","shell.execute_reply.started":"2025-05-16T23:15:42.777734Z","shell.execute_reply":"2025-05-16T23:15:42.791457Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Preprocess Data","metadata":{}},{"cell_type":"code","source":"# Create a label list and mapping dictionary\nlabel_list: list[str] = sorted(train_df[\"primary_label\"].unique())\nlabel_id_list: list[int] = list(range(len(label_list)))\nlabel2id: dict[str, int] = dict(zip(label_list, label_id_list))\nid2label: dict[int, str] = dict(zip(label_id_list, label_list))\n\nprint(f\"Found {len(label_list)} unique species\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-16T23:15:55.400901Z","iopub.execute_input":"2025-05-16T23:15:55.401231Z","iopub.status.idle":"2025-05-16T23:15:55.411138Z","shell.execute_reply.started":"2025-05-16T23:15:55.401202Z","shell.execute_reply":"2025-05-16T23:15:55.410458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a dataframe for preprcessing\nworking_df: pd.DataFrame = train_df[[\"primary_label\", \"rating\", \"filename\"]].copy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T23:27:19.817149Z","iopub.execute_input":"2025-05-14T23:27:19.817473Z","iopub.status.idle":"2025-05-14T23:27:19.831429Z","shell.execute_reply.started":"2025-05-14T23:27:19.817443Z","shell.execute_reply":"2025-05-14T23:27:19.826073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"working_df[\"target\"] = working_df.primary_label.map(label2id)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T23:27:20.792784Z","iopub.execute_input":"2025-05-14T23:27:20.793070Z","iopub.status.idle":"2025-05-14T23:27:20.805921Z","shell.execute_reply.started":"2025-05-14T23:27:20.793046Z","shell.execute_reply":"2025-05-14T23:27:20.801505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"working_df[\"filepath\"] = config.DATA_ROOT + \"/train_audio/\" + working_df.filename","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T23:27:21.041593Z","iopub.execute_input":"2025-05-14T23:27:21.041890Z","iopub.status.idle":"2025-05-14T23:27:21.056250Z","shell.execute_reply.started":"2025-05-14T23:27:21.041866Z","shell.execute_reply":"2025-05-14T23:27:21.051953Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"working_df[\"samplename\"] = working_df.filename.map(lambda x: x.split(\"/\")[0] + \"-\" + x.split(\"/\")[-1].split(\".\")[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T23:27:21.393677Z","iopub.execute_input":"2025-05-14T23:27:21.394022Z","iopub.status.idle":"2025-05-14T23:27:21.431853Z","shell.execute_reply.started":"2025-05-14T23:27:21.393994Z","shell.execute_reply":"2025-05-14T23:27:21.427317Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"working_df[\"class\"] = working_df.primary_label.map(lambda x: species_class_map.get(x, \"Unknown\"))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T23:27:29.353833Z","iopub.execute_input":"2025-05-14T23:27:29.354137Z","iopub.status.idle":"2025-05-14T23:27:29.370594Z","shell.execute_reply.started":"2025-05-14T23:27:29.354112Z","shell.execute_reply":"2025-05-14T23:27:29.366005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"working_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T23:27:29.784025Z","iopub.execute_input":"2025-05-14T23:27:29.784308Z","iopub.status.idle":"2025-05-14T23:27:29.818752Z","shell.execute_reply.started":"2025-05-14T23:27:29.784283Z","shell.execute_reply":"2025-05-14T23:27:29.813111Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"total_samples: int = min(len(working_df), config.N_MAX or len(working_df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T23:27:30.750927Z","iopub.execute_input":"2025-05-14T23:27:30.751220Z","iopub.status.idle":"2025-05-14T23:27:30.760338Z","shell.execute_reply.started":"2025-05-14T23:27:30.751195Z","shell.execute_reply":"2025-05-14T23:27:30.756057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Total samples to process: {total_samples} out of {len(working_df)} available\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T23:27:31.545374Z","iopub.execute_input":"2025-05-14T23:27:31.545668Z","iopub.status.idle":"2025-05-14T23:27:31.555620Z","shell.execute_reply.started":"2025-05-14T23:27:31.545640Z","shell.execute_reply":"2025-05-14T23:27:31.550245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Sample by class\nprint(working_df[\"class\"].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T23:27:31.914958Z","iopub.execute_input":"2025-05-14T23:27:31.915251Z","iopub.status.idle":"2025-05-14T23:27:31.927068Z","shell.execute_reply.started":"2025-05-14T23:27:31.915224Z","shell.execute_reply":"2025-05-14T23:27:31.922897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def audio2melspec(audio_data: np.ndarray) -> np.ndarray:\n    \"\"\"\n    Converts audio data to a normalized Mel spectrogram.\n\n    This function takes a NumPy array representing audio data, calculates its\n    Mel spectrogram, converts it to the decibel scale, and normalizes the\n    values to the range [0, 1]. It handles potential NaN values in the input\n    audio data by filling them with the mean of the non-NaN values.\n\n    Args:\n        audio_data (np.ndarray): A NumPy array containing the audio time series data.\n                                 Expected to be a 1D array of floating-point numbers.\n\n    Returns:\n        np.ndarray: A 2D NumPy array representing the normalized Mel spectrogram.\n                    The shape is (n_mels, time_frames), where n_mels is the\n                    number of Mel bands and time_frames depends on the length\n                    of the audio data and the hop length. The values are\n                    normalized to the range [0, 1].\n    \"\"\"\n    \n    if np.isnan(audio_data).any():\n        mean_signal: float = np.nanmean(audio_data)\n        audio_data: np.ndarray = np.nan_to_num(audio_data, nan=mean_signal)\n\n    mel_spec: np.ndarray = librosa.feature.melspectrogram(\n        y=audio_data,\n        sr=config.FS,\n        n_fft=config.N_FFT,\n        hop_length=config.HOP_LENGTH,\n        n_mels=config.N_MELS,\n        fmin=config.FMIN,\n        fmax=config.FMAX,\n        power=2.0\n    )\n\n    mel_spec_db: np.ndarray = librosa.power_to_db(mel_spec, ref=np.max)\n    mel_spec_norm: np.ndarray = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n\n    return mel_spec_norm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T23:27:32.879644Z","iopub.execute_input":"2025-05-14T23:27:32.879993Z","iopub.status.idle":"2025-05-14T23:27:32.894686Z","shell.execute_reply.started":"2025-05-14T23:27:32.879966Z","shell.execute_reply":"2025-05-14T23:27:32.887427Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Audio Processing","metadata":{}},{"cell_type":"code","source":"# Start audio processing\nprint(f\"{'DEBUG MODE - Processing only 50 samples' if config.DEBUG_MODE else 'FULL MODE - Processing all samples'}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T23:27:34.921356Z","iopub.execute_input":"2025-05-14T23:27:34.921642Z","iopub.status.idle":"2025-05-14T23:27:34.934503Z","shell.execute_reply.started":"2025-05-14T23:27:34.921617Z","shell.execute_reply":"2025-05-14T23:27:34.928451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"start_time: float = time.time()\nall_bird_data: dict[str, np.float32] = {}\nerrors: list[tuple[str, str]] = []\n\nfor i, row in tqdm(working_df.iterrows(), total=total_samples):\n    if config.N_MAX is not None and i >= config.N_MAX:\n        break\n    \n    try:\n        audio_data: np.ndarray\n\n        # Load audio data\n        audio_data, _ = librosa.load(row.filepath, sr=config.FS)\n\n        # Calculates the target number of samples from the specified target length (in seconds) and sampling rate\n        target_samples: int = int(config.TARGET_DURATION * config.FS)\n\n        # If the number of samples of the loaded audio data is shorter than the target number of samples,\n        # the audio is repeated to increase its length.\n        if len(audio_data) < target_samples:\n            n_copy: int = math.ceil(target_samples / len(audio_data))\n            if n_copy > 1:\n                audio_data = np.concatenate([audio_data] * n_copy)\n\n        # The starting index for extracting samples of the target length from the center of the audio data.\n        start_idx: int = max(0, int(len(audio_data) / 2 - target_samples / 2))\n\n        # The end index of the range to extract, without going beyond the range even if the end of the data is reached.        \n        end_idx: int = min(len(audio_data), start_idx + target_samples)\n        \n        # Extract the audio data between the calculated start and end indexes.\n        center_audio: np.ndarray = audio_data[start_idx:end_idx]\n\n        # If the target number of samples is still not reached after extraction\n        # (for example because the original audio data was extremely short),\n        # the end of the extracted audio data is padded with 0 until the target number of samples is reached.\n        if len(center_audio) < target_samples:\n            center_audio: np.ndarray = np.pad(\n                center_audio,\n                (0, target_samples - len(center_audio)),\n                mode=\"constant\"\n            )\n\n        # Calculate mel spectrogram\n        mel_spec: np.ndarray = audio2melspec(center_audio)\n\n        # Resize the calculated mel spectrogram if it differs from the set target shape.\n        if mel_spec.shape != config.TARGET_SHAPE:\n            mel_spec: np.ndarray = cv2.resize(mel_spec, config.TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n\n        # Store the processed mel spectrograms (NumPy Arrays).\n        all_bird_data[row.samplename] = mel_spec.astype(np.float32)\n        \n    except Exception as e:\n        print(f\"Error processing {row.filepath}: {e}\")\n        errors.append((row.filepath, str(e)))\n\nend_time: float = time.time()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-14T23:29:03.880260Z","iopub.execute_input":"2025-05-14T23:29:03.880584Z","iopub.status.idle":"2025-05-14T23:29:03.952843Z","shell.execute_reply.started":"2025-05-14T23:29:03.880556Z","shell.execute_reply":"2025-05-14T23:29:03.949510Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Processing completed in {end_time - start_time:.2f} seconds\")\nprint(f\"Successfully processed {len(all_bird_data)} files out of {total_samples} total\")\nprint(f\"Failed to process {len(errors)} files\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"samples: list = []\ndisplayed_classes = set()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"max_samples: int = min(4, len(all_bird_data))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i, row in working_df.iterrows():\n    if i >= (config.N_MAX or len(working_df)):\n        break\n        \n    if row[\"samplename\"] in all_bird_data:\n        if config.DEBUG_MODE:\n            if row[\"class\"] not in displayed_classes:\n                samples.append((row[\"samplename\"], row[\"class\"], row[\"primary_label\"]))\n                displayed_classes.add(row[\"class\"])\n        else:\n            if row[\"class\"] not in displayed_classes:\n                samples.append((row[\"samplename\"], row[\"class\"], row[\"primary_label\"]))\n                displayed_classes.add(row[\"class\"])\n        \n        if len(samples) >= max_samples:  \n            break","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Visualization","metadata":{}},{"cell_type":"code","source":"if samples:\n    plt.figure(figsize=(16, 12))\n    \n    for i, (samplename, class_name, species) in enumerate(samples):\n        plt.subplot(2, 2, i+1)\n        plt.imshow(all_bird_data[samplename], aspect=\"auto\", origin=\"lower\", cmap=\"viridis\")\n        plt.title(f\"{class_name}: {species}\")\n        plt.colorbar(format=\"%+2.0f dB\")\n    \n    plt.tight_layout()\n    debug_note = \"debug_\" if config.DEBUG_MODE else \"\"\n    plt.savefig(f\"{debug_note}melspec_examples.png\")\n    plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save Data","metadata":{}},{"cell_type":"code","source":"output_path: str = f\"{config.OUTPUT_DIR}/birdclef2025_melspec_{int(config.TARGET_DURATION)}sec_{config.TARGET_SHAPE[0]}_{config.TARGET_SHAPE[1]}.npy\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.save(output_path, all_bird_data)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}