{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **BirdCLEF 2025 Data Preprocessing Notebook**\nThis notebook is adopted from [this notebook](https://www.kaggle.com/code/kadircandrisolu/transforming-audio-to-mel-spec-birdclef-25), with the following key difference:\n\nIn this work, LMDB (Lightning Memory-Mapped Database) is used to store the spectrograms instead of .npy files. This change provides several benefits:\n\nBenefits of Using LMDB over .npy for Large Datasets:\n\n\n* Efficient Storage: LMDB provides a compact, memory-mapped storage format, which is more efficient for large datasets compared to .npy. It avoids duplicating data and uses less disk space.\n\n* Faster Access: LMDB allows fast random access to data, making it highly suitable for large-scale deep learning applications where frequent data loading is required during training.\n\n* Memory Mapping: With memory-mapped data, LMDB allows accessing large datasets **without** loading them **entirely into memory**, preventing memory overflows and optimizing resource usage.\n\n* Atomic Transactions: LMDB supports atomic operations for data insertion and updates, which ensures data consistency and avoids corruption during concurrent read/write operations.\n\n* Scalability: LMDB can handle larger datasets than .npy files, as the database can grow dynamically without performance degradation, especially in distributed or parallel training scenarios.\n\nEfficient Use of Disk Space: LMDB stores data in a compressed, binary format, making it more efficient than storing individual .npy files for each sample, especially when working with large volumes of data.\n","metadata":{}},{"cell_type":"code","source":"!pip install lmdb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T10:26:07.410281Z","iopub.execute_input":"2025-03-24T10:26:07.410734Z","iopub.status.idle":"2025-03-24T10:26:13.935955Z","shell.execute_reply.started":"2025-03-24T10:26:07.410699Z","shell.execute_reply":"2025-03-24T10:26:13.934516Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport io\nimport cv2\nimport math\nimport time\nimport librosa\nimport pandas as pd\nimport numpy as np\nimport lmdb\n\nfrom tqdm.notebook import tqdm\n\nimport torch\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T10:26:13.937717Z","iopub.execute_input":"2025-03-24T10:26:13.938288Z","iopub.status.idle":"2025-03-24T10:26:19.440852Z","shell.execute_reply.started":"2025-03-24T10:26:13.938243Z","shell.execute_reply":"2025-03-24T10:26:19.439493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Config:\n \n    DEBUG_MODE = True\n    \n    OUTPUT_DIR = '/kaggle/working/'\n    DATA_ROOT = '/kaggle/input/birdclef-2025'\n    FS = 32000\n    \n    # Mel spectrogram parameters\n    N_FFT = 1024\n    HOP_LENGTH = 512\n    N_MELS = 128\n    FMIN = 50\n    FMAX = 14000\n    \n    TARGET_DURATION = 5.0\n    TARGET_SHAPE = (256, 256)  \n    \n    N_MAX = 50 if DEBUG_MODE else None  \n\nconfig = Config()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T10:26:19.443252Z","iopub.execute_input":"2025-03-24T10:26:19.443896Z","iopub.status.idle":"2025-03-24T10:26:19.450209Z","shell.execute_reply.started":"2025-03-24T10:26:19.443857Z","shell.execute_reply":"2025-03-24T10:26:19.449105Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Debug mode: {'ON' if config.DEBUG_MODE else 'OFF'}\")\nprint(f\"Max samples to process: {config.N_MAX if config.N_MAX is not None else 'ALL'}\")\n\nprint(\"Loading taxonomy data...\")\ntaxonomy_df = pd.read_csv(f'{config.DATA_ROOT}/taxonomy.csv')\nspecies_class_map = dict(zip(taxonomy_df['primary_label'], taxonomy_df['class_name']))\n\nprint(\"Loading training metadata...\")\ntrain_df = pd.read_csv(f'{config.DATA_ROOT}/train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T10:26:19.452127Z","iopub.execute_input":"2025-03-24T10:26:19.452518Z","iopub.status.idle":"2025-03-24T10:26:19.743158Z","shell.execute_reply.started":"2025-03-24T10:26:19.452488Z","shell.execute_reply":"2025-03-24T10:26:19.742104Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_list = sorted(train_df['primary_label'].unique())\nlabel_id_list = list(range(len(label_list)))\nlabel2id = dict(zip(label_list, label_id_list))\nid2label = dict(zip(label_id_list, label_list))\n\nprint(f'Found {len(label_list)} unique species')\nworking_df = train_df[['primary_label', 'rating', 'filename']].copy()\nworking_df['target'] = working_df.primary_label.map(label2id)\nworking_df['filepath'] = config.DATA_ROOT + '/train_audio/' + working_df.filename\nworking_df['samplename'] = working_df.filename.map(lambda x: x.split('/')[0] + '-' + x.split('/')[-1].split('.')[0])\nworking_df['class'] = working_df.primary_label.map(lambda x: species_class_map.get(x, 'Unknown'))\ntotal_samples = min(len(working_df), config.N_MAX or len(working_df))\nprint(f'Total samples to process: {total_samples} out of {len(working_df)} available')\nprint(f'Samples by class:')\nprint(working_df['class'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T10:26:19.744212Z","iopub.execute_input":"2025-03-24T10:26:19.744658Z","iopub.status.idle":"2025-03-24T10:26:19.824223Z","shell.execute_reply.started":"2025-03-24T10:26:19.744611Z","shell.execute_reply":"2025-03-24T10:26:19.823023Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def audio2melspec(audio_data):\n    if np.isnan(audio_data).any():\n        mean_signal = np.nanmean(audio_data)\n        audio_data = np.nan_to_num(audio_data, nan=mean_signal)\n\n    mel_spec = librosa.feature.melspectrogram(\n        y=audio_data,\n        sr=config.FS,\n        n_fft=config.N_FFT,\n        hop_length=config.HOP_LENGTH,\n        n_mels=config.N_MELS,\n        fmin=config.FMIN,\n        fmax=config.FMAX,\n        power=2.0\n    )\n\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n    mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n    \n    return mel_spec_norm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T10:26:19.825319Z","iopub.execute_input":"2025-03-24T10:26:19.825711Z","iopub.status.idle":"2025-03-24T10:26:19.832327Z","shell.execute_reply.started":"2025-03-24T10:26:19.825681Z","shell.execute_reply":"2025-03-24T10:26:19.830853Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Starting audio processing...\")\nprint(f\"{'DEBUG MODE - Processing only 50 samples' if config.DEBUG_MODE else 'FULL MODE - Processing all samples'}\")\nstart_time = time.time()\n\nall_bird_data = {}\nerrors = []\n\nenv = lmdb.open('./mel_specs.lmdb', map_size=10 * 1024**3 , writemap=True)\n\nwith env.begin(write=True) as txn:\n    \n    for i, row in tqdm(working_df.iterrows(), total=total_samples):\n        if config.N_MAX is not None and i >= config.N_MAX:\n            break\n        \n        try:\n            audio_data, _ = librosa.load(row.filepath, sr=config.FS)\n    \n            target_samples = int(config.TARGET_DURATION * config.FS)\n    \n            if len(audio_data) < target_samples:\n                n_copy = math.ceil(target_samples / len(audio_data))\n                if n_copy > 1:\n                    audio_data = np.concatenate([audio_data] * n_copy)\n    \n            start_idx = max(0, int(len(audio_data) / 2 - target_samples / 2))\n            end_idx = min(len(audio_data), start_idx + target_samples)\n            center_audio = audio_data[start_idx:end_idx]\n    \n            if len(center_audio) < target_samples:\n                center_audio = np.pad(center_audio, \n                                     (0, target_samples - len(center_audio)), \n                                     mode='constant')\n    \n            mel_spec = audio2melspec(center_audio)\n    \n            if mel_spec.shape != config.TARGET_SHAPE:\n                mel_spec = cv2.resize(mel_spec, config.TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n    \n            all_bird_data[row.samplename] = mel_spec.astype(np.float32)\n    \n        \n            key = row.filename.encode(\"ascii\")\n            buffer = io.BytesIO()\n            np.savez_compressed(buffer, mel=mel_spec.astype(np.float32))\n            txn.put(key, buffer.getvalue())  # Store compressed data\n            \n        except Exception as e:\n            print(f\"Error processing {row.filepath}: {e}\")\n            errors.append((row.filepath, str(e)))\n\nend_time = time.time()\nprint(f\"Processing completed in {end_time - start_time:.2f} seconds\")\nprint(f\"Successfully processed {len(all_bird_data)} files out of {total_samples} total\")\nprint(f\"Failed to process {len(errors)} files\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T10:26:21.545058Z","iopub.execute_input":"2025-03-24T10:26:21.545487Z","iopub.status.idle":"2025-03-24T10:26:47.391593Z","shell.execute_reply.started":"2025-03-24T10:26:21.545455Z","shell.execute_reply":"2025-03-24T10:26:47.390405Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Read File from LMDB","metadata":{}},{"cell_type":"code","source":"def read_mel_from_lmdb(lmdb_path, filename):\n    env = lmdb.open(lmdb_path, readonly=True, lock=False)\n    \n    with env.begin() as txn:\n        key = filename.encode(\"ascii\")  # Convert filename to bytes\n        value = txn.get(key)\n\n        if value is None:\n            print(f\"File {filename} not found in LMDB.\")\n            return None\n        \n        buffer = io.BytesIO(value)  # Load from bytes\n        mel_data = np.load(buffer)['mel']  # Extract mel spectrogram\n        \n        return mel_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T10:26:47.393028Z","iopub.execute_input":"2025-03-24T10:26:47.393634Z","iopub.status.idle":"2025-03-24T10:26:47.399071Z","shell.execute_reply.started":"2025-03-24T10:26:47.393593Z","shell.execute_reply":"2025-03-24T10:26:47.398091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nmel_data = read_mel_from_lmdb('./mel_specs.lmdb',working_df.iloc[0]['filename'])\n\nplt.imshow(mel_data, aspect='auto', origin='lower', cmap='viridis')\nplt.title(f\"Mel Spectrogram\")\nplt.colorbar(format='%+2.0f dB')\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-24T10:26:47.401320Z","iopub.execute_input":"2025-03-24T10:26:47.401646Z","iopub.status.idle":"2025-03-24T10:26:47.978727Z","shell.execute_reply.started":"2025-03-24T10:26:47.401619Z","shell.execute_reply":"2025-03-24T10:26:47.976791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}