{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91844,"databundleVersionId":11361821,"sourceType":"competition"},{"sourceId":11782099,"sourceType":"datasetVersion","datasetId":7397213},{"sourceId":234320181,"sourceType":"kernelVersion"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport librosa\nimport cv2\nimport pickle\n#pytorch関連\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torchinfo import summary\nimport torchvision.transforms as transforms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T12:55:15.090923Z","iopub.execute_input":"2025-05-24T12:55:15.091333Z","iopub.status.idle":"2025-05-24T12:55:15.097941Z","shell.execute_reply.started":"2025-05-24T12:55:15.091306Z","shell.execute_reply":"2025-05-24T12:55:15.096537Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 実験設定","metadata":{"execution":{"iopub.status.busy":"2025-05-12T05:39:36.153223Z","iopub.execute_input":"2025-05-12T05:39:36.155596Z","iopub.status.idle":"2025-05-12T05:39:36.163071Z","shell.execute_reply.started":"2025-05-12T05:39:36.155540Z","shell.execute_reply":"2025-05-12T05:39:36.161435Z"}}},{"cell_type":"code","source":"class CFG:\n    # 音声データの変換関係のパラメータ\n    FS = 32000\n    N_FFT = 1024\n    HOP_LENGTH = 512\n    N_MELS = 128\n    FMIN = 50\n    FMAX = 14000\n    TARGET_DURATION = 5\n    TARGET_SHAPE = (256, 256)\n\n    extract_human_voice = False\n    debug = False","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T12:46:34.768410Z","iopub.execute_input":"2025-05-24T12:46:34.768898Z","iopub.status.idle":"2025-05-24T12:46:34.775111Z","shell.execute_reply.started":"2025-05-24T12:46:34.768866Z","shell.execute_reply":"2025-05-24T12:46:34.773773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cfg = CFG()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T12:48:46.696695Z","iopub.execute_input":"2025-05-24T12:48:46.697075Z","iopub.status.idle":"2025-05-24T12:48:46.702048Z","shell.execute_reply.started":"2025-05-24T12:48:46.697051Z","shell.execute_reply":"2025-05-24T12:48:46.701067Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# データ作成","metadata":{}},{"cell_type":"markdown","source":"## メルスペクトログラムへの変換","metadata":{}},{"cell_type":"code","source":"def exclude_human_voice(audio_data: np.array, voice_segments: list[dict], cfg: CFG) -> np.ndarray:\n    \"\"\"train_audioに含まれる人の声を削除する.\"\"\"\n    # 除外区間の前後をつなげていく\n    include_segments = []\n    prev_end_sample = 0\n    \n    for seg in voice_segments:\n        start_sample = int(seg['start'] * cfg.FS)\n        end_sample = int(seg['end'] * cfg.FS)\n    \n        if start_sample > prev_end_sample:\n            include_segments.append(audio_data[prev_end_sample:start_sample])\n    \n        prev_end_sample = end_sample\n    \n    # 最後に残りがあれば追加\n    if prev_end_sample < len(audio_data):\n        include_segments.append(audio_data[prev_end_sample:])\n    \n    # すべて連結\n    cleaned_audio = np.concatenate(include_segments)\n    \n    return cleaned_audio\n\ndef audio2melspec(audio_data: np.ndarray, cfg: CFG) -> np.ndarray:\n    \"\"\"音声データをデシベル単位のメルスペクトログラムに変換し、正規化(Min-Max正規化)する\"\"\"\n\n    if np.isnan(audio_data).any():\n        mean_signal = np.nanmean(audio_data)\n        audio_data = np.nan_to_num(audio_data, nan=mean_signal)\n    \n    mel_spec = librosa.feature.melspectrogram(\n        y=audio_data,\n        sr=cfg.FS,\n        n_fft=cfg.N_FFT,\n        n_mels=cfg.N_MELS,\n        fmin=cfg.FMIN,\n        fmax=cfg.FMAX,\n        hop_length=cfg.HOP_LENGTH\n    )\n\n    mel_spec_db = librosa.power_to_db(mel_spec, ref=np.max)\n    mel_spec_norm = (mel_spec_db - mel_spec_db.min()) / (mel_spec_db.max() - mel_spec_db.min() + 1e-8)\n\n    return mel_spec_norm\n\ndef preprocess_audio_file(audio_path: str, voice_segments: dict, cfg: CFG) -> np.ndarray:\n    \"\"\"音声データを読み込み、真ん中の5秒間を正規化したメルスペクトログラムに変換する(db単位)\"\"\"\n    \n    try:\n        # 読み込み\n        audio_data, _ = librosa.load(audio_path, sr=cfg.FS)\n        if cfg.extract_human_voice:\n            # 人間の声を除く\n            audio_data = exclude_human_voice(audio_data, voice_segments, cfg)\n        \n        target_samples = int(cfg.TARGET_DURATION * cfg.FS)\n        start_idx = max(0, int(len(audio_data) / 2 - target_samples / 2))\n        end_idx = min(len(audio_data), start_idx + target_samples)\n        center_audio = audio_data[start_idx: end_idx]\n\n        if len(center_audio) < target_samples:\n            center_audio = np.pad(\n                center_audio,\n                pad_width = (0, target_samples - len(center_audio)),\n                mode = 'constant'\n            )\n\n        mel_spec = audio2melspec(center_audio, cfg)\n\n        if mel_spec.shape != cfg.TARGET_SHAPE:\n            mel_spec = cv2.resize(mel_spec, cfg.TARGET_SHAPE, interpolation=cv2.INTER_LINEAR)\n\n        return mel_spec.astype(np.float32)\n\n    except Exception as e:\n        print(f\"Error processing {audio_path}: {e}\")\n        return None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T12:59:43.826559Z","iopub.execute_input":"2025-05-24T12:59:43.828917Z","iopub.status.idle":"2025-05-24T12:59:43.850263Z","shell.execute_reply.started":"2025-05-24T12:59:43.828858Z","shell.execute_reply":"2025-05-24T12:59:43.848831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 実際にメルスペクトログラム変換していく\ntrain_csv = pd.read_csv(\"/kaggle/input/birdclef-2025/train.csv\")\nwith open(\"/kaggle/input/bc25-separation-voice-from-data/train_voice_data.pkl\", \"rb\") as f:\n    train_voice_data = pickle.load(f)\n\nimages = []\nlabels = []\n\ncnt_for_debug = 0\n\nfor _, row in tqdm(train_csv.iterrows(), total=28564):\n    audio_path = os.path.join(\"/kaggle/input/birdclef-2025/train_audio\", row[\"filename\"])\n    voice_segments = train_voice_data.get(audio_path, [])\n    \n    image = preprocess_audio_file(\n        audio_path=audio_path,\n        voice_segments=voice_segments,\n        cfg=cfg\n    )\n    images.append(image)\n    labels.append(row['primary_label'])\n\n    if cfg.debug:\n        cnt_for_debug += 1\n        if cnt_for_debug >= 10:\n            break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T13:04:28.223233Z","iopub.execute_input":"2025-05-24T13:04:28.224107Z","iopub.status.idle":"2025-05-24T13:04:29.651065Z","shell.execute_reply.started":"2025-05-24T13:04:28.224063Z","shell.execute_reply":"2025-05-24T13:04:29.650019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(np.array(images).shape)\nprint(np.array(labels).shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-24T13:04:31.333772Z","iopub.execute_input":"2025-05-24T13:04:31.334764Z","iopub.status.idle":"2025-05-24T13:04:31.341544Z","shell.execute_reply.started":"2025-05-24T13:04:31.334728Z","shell.execute_reply":"2025-05-24T13:04:31.340436Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"np.savez(\"/kaggle/working/my_dataset_exclude_human_voice.npz\", images=np.array(images), labels=np.array(labels))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-12T10:10:48.077513Z","iopub.execute_input":"2025-05-12T10:10:48.078210Z","iopub.status.idle":"2025-05-12T10:11:19.222316Z","shell.execute_reply.started":"2025-05-12T10:10:48.078169Z","shell.execute_reply":"2025-05-12T10:11:19.221065Z"}},"outputs":[],"execution_count":null}]}