{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"}],"dockerImageVersionId":31011,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"可视化每个类别的数量","metadata":{}},{"cell_type":"code","source":"import os\nimport matplotlib.pyplot as plt\n\n# Function to visualize the file count in each subfolder\ndef visualize_file_counts(parent_folder):\n    # Get list of subfolders\n    subfolders = [f.path for f in os.scandir(parent_folder) if f.is_dir()]\n    \n    # Count the files in each subfolder\n    folder_counts = []\n    for subfolder in subfolders:\n        file_count = len([f for f in os.scandir(subfolder) if f.is_file()])\n        folder_counts.append(file_count)\n    \n    # Plotting\n    plt.figure(figsize=(20, 12))\n    plt.barh(subfolders, folder_counts, color='skyblue')\n    plt.xlabel('Number of Files')\n    plt.ylabel('Subfolders')\n    plt.title('Number of Files in Each Subfolder')\n    plt.tight_layout()\n    plt.show()\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T04:49:14.904374Z","iopub.execute_input":"2025-04-22T04:49:14.904992Z","iopub.status.idle":"2025-04-22T04:49:14.916788Z","shell.execute_reply.started":"2025-04-22T04:49:14.904958Z","shell.execute_reply":"2025-04-22T04:49:14.915897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Example usage (replace with your folder path)\nparent_folder = '/kaggle/input/birdclef-2025/train_audio'  # Replace with your folder path\nvisualize_file_counts(parent_folder)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T03:07:31.650807Z","iopub.execute_input":"2025-04-22T03:07:31.651099Z","iopub.status.idle":"2025-04-22T03:07:49.439308Z","shell.execute_reply.started":"2025-04-22T03:07:31.651074Z","shell.execute_reply":"2025-04-22T03:07:49.438438Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport math\nimport time\nimport librosa\nimport pandas as pd\nimport numpy as np\nfrom tqdm.notebook import tqdm\n\nimport torch\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T04:49:19.163905Z","iopub.execute_input":"2025-04-22T04:49:19.164910Z","iopub.status.idle":"2025-04-22T04:49:24.277066Z","shell.execute_reply.started":"2025-04-22T04:49:19.164872Z","shell.execute_reply":"2025-04-22T04:49:24.276303Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"配置参数设置","metadata":{}},{"cell_type":"code","source":"class Config:\n \n    DEBUG_MODE = False\n    \n    OUTPUT_DIR = '/kaggle/working/'\n    DATA_ROOT = '/kaggle/input/birdclef-2025'\n    FS = 32000\n    \n    # Mel spectrogram parameters\n    N_FFT = 1024\n    HOP_LENGTH = 512\n    WIN_LENGTH = 1024\n    N_MELS = 128\n    FMIN = 20\n    FMAX = 16000\n    \n    TARGET_DURATION = 5.0\n    TARGET_SHAPE = (256, 256)  \n    \n    N_MAX = 50 if DEBUG_MODE else None  \n\nconfig = Config()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T04:50:34.746572Z","iopub.execute_input":"2025-04-22T04:50:34.747047Z","iopub.status.idle":"2025-04-22T04:50:34.751785Z","shell.execute_reply.started":"2025-04-22T04:50:34.747019Z","shell.execute_reply":"2025-04-22T04:50:34.750974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(f\"Debug mode: {'ON' if config.DEBUG_MODE else 'OFF'}\")\nprint(f\"Max samples to process: {config.N_MAX if config.N_MAX is not None else 'ALL'}\")\n\nprint(\"Loading taxonomy data...\")\ntaxonomy_df = pd.read_csv(f'{config.DATA_ROOT}/taxonomy.csv')\nspecies_class_map = dict(zip(taxonomy_df['primary_label'], taxonomy_df['class_name']))\n\nprint(\"Loading training metadata...\")\ntrain_df = pd.read_csv(f'{config.DATA_ROOT}/train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T04:50:40.774435Z","iopub.execute_input":"2025-04-22T04:50:40.774983Z","iopub.status.idle":"2025-04-22T04:50:40.945058Z","shell.execute_reply.started":"2025-04-22T04:50:40.774961Z","shell.execute_reply":"2025-04-22T04:50:40.944227Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"taxonomy_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T04:35:48.869139Z","iopub.execute_input":"2025-04-22T04:35:48.869493Z","iopub.status.idle":"2025-04-22T04:35:48.891173Z","shell.execute_reply.started":"2025-04-22T04:35:48.869470Z","shell.execute_reply":"2025-04-22T04:35:48.890414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T04:36:07.724361Z","iopub.execute_input":"2025-04-22T04:36:07.724964Z","iopub.status.idle":"2025-04-22T04:36:07.739118Z","shell.execute_reply.started":"2025-04-22T04:36:07.724942Z","shell.execute_reply":"2025-04-22T04:36:07.738220Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"查看样本数据量","metadata":{}},{"cell_type":"code","source":"label_list = sorted(train_df['primary_label'].unique())\nlabel_id_list = list(range(len(label_list)))\nlabel2id = dict(zip(label_list, label_id_list))\nid2label = dict(zip(label_id_list, label_list))\n\nprint(f'Found {len(label_list)} unique species')\nworking_df = train_df[['primary_label', 'rating', 'filename']].copy()\nworking_df['target'] = working_df.primary_label.map(label2id)\nworking_df['filepath'] = config.DATA_ROOT + '/train_audio/' + working_df.filename\nworking_df['samplename'] = working_df.filename.map(lambda x: x.split('/')[0] + '-' + x.split('/')[-1].split('.')[0])\nworking_df['class'] = working_df.primary_label.map(lambda x: species_class_map.get(x, 'Unknown'))\ntotal_samples = min(len(working_df), config.N_MAX or len(working_df))\nprint(f'Total samples to process: {total_samples} out of {len(working_df)} available')\nprint(f'Samples by class:')\nprint(working_df['class'].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T04:50:46.285143Z","iopub.execute_input":"2025-04-22T04:50:46.285910Z","iopub.status.idle":"2025-04-22T04:50:46.341770Z","shell.execute_reply.started":"2025-04-22T04:50:46.285873Z","shell.execute_reply":"2025-04-22T04:50:46.341193Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"working_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T06:51:37.305823Z","iopub.execute_input":"2025-04-22T06:51:37.306158Z","iopub.status.idle":"2025-04-22T06:51:37.330737Z","shell.execute_reply.started":"2025-04-22T06:51:37.306106Z","shell.execute_reply":"2025-04-22T06:51:37.330144Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"定义函数：audio2melspec函数，音频转梅尔频谱","metadata":{}},{"cell_type":"code","source":"def audio2melspec(audio_data):\n    if np.isnan(audio_data).any():\n        mean_signal = np.nanmean(audio_data)\n        audio_data = np.nan_to_num(audio_data, nan=mean_signal)\n\n    mel_spec = librosa.feature.melspectrogram(\n        y=audio_data,\n        sr=config.FS,\n        n_fft=config.N_FFT,\n        hop_length=config.HOP_LENGTH,\n        win_length=config.WIN_LENGTH,\n        n_mels=config.N_MELS,\n        fmin=config.FMIN,\n        fmax=config.FMAX,\n        power=2.0\n    )\n\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n    mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n    \n    return mel_spec_norm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T04:50:50.483755Z","iopub.execute_input":"2025-04-22T04:50:50.484537Z","iopub.status.idle":"2025-04-22T04:50:50.489555Z","shell.execute_reply.started":"2025-04-22T04:50:50.484512Z","shell.execute_reply":"2025-04-22T04:50:50.488915Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"开始转换，对所有样本（DEBUG_MODE=False）","metadata":{}},{"cell_type":"code","source":"print(\"Starting audio processing...\")\nprint(f\"{'DEBUG MODE - Processing only 50 samples' if config.DEBUG_MODE else 'FULL MODE - Processing all samples'}\")\nstart_time = time.time()\n\nall_bird_data = {}\nerrors = []\n\nfor i, row in tqdm(working_df.iterrows(), total=total_samples):\n    if config.N_MAX is not None and i >= config.N_MAX:\n        break\n    \n    try:\n        audio_data, _ = librosa.load(row.filepath, sr=config.FS)\n\n        target_samples = int(config.TARGET_DURATION * config.FS)\n\n        if len(audio_data) < target_samples:\n            n_copy = math.ceil(target_samples / len(audio_data))\n            if n_copy > 1:\n                audio_data = np.concatenate([audio_data] * n_copy)\n\n        start_idx = max(0, int(len(audio_data) / 2 - target_samples / 2))\n        end_idx = min(len(audio_data), start_idx + target_samples)\n        center_audio = audio_data[start_idx:end_idx]\n\n        if len(center_audio) < target_samples:\n            center_audio = np.pad(center_audio, \n                                 (0, target_samples - len(center_audio)), \n                                 mode='constant')\n\n        mel_spec = audio2melspec(center_audio)\n\n        if mel_spec.shape != config.TARGET_SHAPE:\n            mel_spec = cv2.resize(mel_spec, config.TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n\n        all_bird_data[row.samplename] = mel_spec.astype(np.float32)\n        \n    except Exception as e:\n        print(f\"Error processing {row.filepath}: {e}\")\n        errors.append((row.filepath, str(e)))\n\nend_time = time.time()\nprint(f\"Processing completed in {end_time - start_time:.2f} seconds\")\nprint(f\"Successfully processed {len(all_bird_data)} files out of {total_samples} total\")\nprint(f\"Failed to process {len(errors)} files\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T04:50:53.090732Z","iopub.execute_input":"2025-04-22T04:50:53.091404Z","iopub.status.idle":"2025-04-22T05:18:10.225831Z","shell.execute_reply.started":"2025-04-22T04:50:53.091380Z","shell.execute_reply":"2025-04-22T05:18:10.224924Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"保存为npy文件待使用","metadata":{}},{"cell_type":"code","source":"output_filename = os.path.join(config.OUTPUT_DIR, \"birdclef2025_melspec_5sec_256_256.npy\")\nnp.save(output_filename, all_bird_data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T05:21:35.668694Z","iopub.execute_input":"2025-04-22T05:21:35.669421Z","iopub.status.idle":"2025-04-22T05:21:56.907052Z","shell.execute_reply.started":"2025-04-22T05:21:35.669395Z","shell.execute_reply":"2025-04-22T05:21:56.906125Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"All_bird_data = np.load('/kaggle/working/birdclef2025_melspec_5sec_256_256.npy',allow_pickle=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T06:33:35.419970Z","iopub.execute_input":"2025-04-22T06:33:35.420246Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"可视化转换后的频谱图","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nsamples = []\ndisplayed_classes = set()\n\nmax_samples = min(4, len(all_bird_data))\n\nfor i, row in working_df.iterrows():\n    if i >= (config.N_MAX or len(working_df)):\n        break\n        \n    if row['samplename'] in all_bird_data:\n        if config.DEBUG_MODE:\n            if row['class'] not in displayed_classes:\n                samples.append((row['samplename'], row['class'], row['primary_label']))\n                displayed_classes.add(row['class'])\n        else:\n            if row['class'] not in displayed_classes:\n                samples.append((row['samplename'], row['class'], row['primary_label']))\n                displayed_classes.add(row['class'])\n        \n        if len(samples) >= max_samples:  \n            break\n\nif samples:\n    plt.figure(figsize=(16, 12))\n    \n    for i, (samplename, class_name, species) in enumerate(samples):\n        plt.subplot(2, 2, i+1)\n        plt.imshow(all_bird_data[samplename], aspect='auto', origin='lower', cmap='viridis')\n        plt.title(f\"{class_name}: {species}\")\n        plt.colorbar(format='%+2.0f dB')\n    \n    plt.tight_layout()\n    debug_note = \"debug_\" if config.DEBUG_MODE else \"\"\n    plt.savefig(f'{debug_note}melspec_examples.png')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-22T06:41:57.862908Z","iopub.execute_input":"2025-04-22T06:41:57.863283Z","iopub.status.idle":"2025-04-22T06:42:00.876392Z","shell.execute_reply.started":"2025-04-22T06:41:57.863261Z","shell.execute_reply":"2025-04-22T06:42:00.875350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}