{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":25954,"databundleVersionId":2091745,"sourceType":"competition"},{"sourceId":177518,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":151224,"modelId":173697}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport librosa as lb\nimport warnings\nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib as mpl\ncmap = mpl.colormaps['coolwarm']\n\n%matplotlib inline\nwarnings.filterwarnings(\"ignore\")\n\nfrom pathlib import Path\nfrom tqdm.notebook import tqdm\nfrom PIL import Image\nfrom sklearn.utils import shuffle\nimport cv2\nimport concurrent.futures\nfrom os import listdir\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torchvision import models\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import DataLoader, random_split, Dataset\nfrom torchvision.transforms import transforms\nimport torch.nn.functional as F\n\ndevice = 'cuda:0' if torch.cuda.is_available() else 'cpu'\nprint(device)\n\nNUM_CLASSES = 397\nSAMPLE_RATE = 32000\nSIGNAL_LENGTH = 5\nSPEC_SHAPE = (48, 128)\nFMIN = 20\nFMAX = 12000\nTHRESHOLD = 0.2\nN_FFT = 1024","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:24:27.647553Z","iopub.execute_input":"2024-11-24T23:24:27.647952Z","iopub.status.idle":"2024-11-24T23:24:33.410402Z","shell.execute_reply.started":"2024-11-24T23:24:27.647902Z","shell.execute_reply":"2024-11-24T23:24:33.409178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/birdclef-2021/train_metadata.csv\")\n\nLABEL_IDS = {label: label_id for label_id,label in enumerate(sorted(df_train[\"primary_label\"].unique()))}\nINV_LABEL_IDS = {val: key for key,val in LABEL_IDS.items()}\nlen(INV_LABEL_IDS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:24:33.412074Z","iopub.execute_input":"2024-11-24T23:24:33.412556Z","iopub.status.idle":"2024-11-24T23:24:33.999892Z","shell.execute_reply.started":"2024-11-24T23:24:33.412523Z","shell.execute_reply":"2024-11-24T23:24:33.998743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# class SEBlock(nn.Module):\n#     def __init__(self, in_channels, reduction=16):\n#         super(SEBlock, self).__init__()\n#         self.fc1 = nn.Linear(in_channels, in_channels // reduction)\n#         self.fc2 = nn.Linear(in_channels // reduction, in_channels)\n#         self.sigmoid = nn.Sigmoid()\n\n#     def forward(self, x):\n#         # Сжимаем каналы\n#         b, c, _, _ = x.size()\n#         squeeze = torch.mean(x, dim=[2, 3], keepdim=False)\n#         excitation = self.fc2(torch.relu(self.fc1(squeeze)))\n#         excitation = self.sigmoid(excitation).unsqueeze(2).unsqueeze(3)\n#         return x * excitation\n\n# class BirdSoundClassifier(nn.Module):\n#     def __init__(self, num_classes=397):\n#         super(BirdSoundClassifier, self).__init__()\n\n#         # Свёрточные блоки\n#         self.conv1 = nn.Conv2d(3, 64, kernel_size=3, padding=1)\n#         self.bn1 = nn.BatchNorm2d(64)\n#         self.pool1 = nn.MaxPool2d(kernel_size=2, stride=2)\n\n#         self.conv2 = nn.Conv2d(64, 128, kernel_size=3, padding=1)\n#         self.bn2 = nn.BatchNorm2d(128)\n#         self.pool2 = nn.MaxPool2d(kernel_size=2, stride=2)\n\n#         self.conv3 = nn.Conv2d(128, 256, kernel_size=3, padding=1)\n#         self.bn3 = nn.BatchNorm2d(256)\n#         self.pool3 = nn.MaxPool2d(kernel_size=2, stride=2)\n\n#         # self.conv4 = nn.Conv2d(256, 512, kernel_size=3, padding=1)\n#         # self.bn4 = nn.BatchNorm2d(512)\n#         # self.pool4 = nn.MaxPool2d(kernel_size=2, stride=2)\n\n#         self.se_block = SEBlock(256)\n\n#         # Глобальное усреднение\n#         self.global_avg_pool = nn.AdaptiveAvgPool2d(1)  # Сжимает до одного пикселя\n\n#         # Полносвязные слои\n#         self.fc1 = nn.Linear(256, 512)\n#         #self.fc2 = nn.Linear(512, 512)\n#         self.fc3 = nn.Linear(512, num_classes)  # num_classes — это количество классов\n\n#         # Dropout для регуляризации\n#         self.dropout = nn.Dropout(0.5)\n\n#     def forward(self, x):\n#         # Применяем свёрточные слои с активациями, нормализацией и пулингом\n#         x = self.pool1(torch.relu(self.bn1(self.conv1(x))))\n#         x = self.pool2(torch.relu(self.bn2(self.conv2(x))))\n#         x = self.pool3(torch.relu(self.bn3(self.conv3(x))))\n#         #x = self.pool4(torch.relu(self.bn4(self.conv4(x))))\n\n#         # Применяем слой внимания\n#         x = self.se_block(x)\n\n#         # Глобальное усреднение\n#         x = self.global_avg_pool(x)\n\n#         # Преобразуем выход в вектор\n#         x = torch.flatten(x, 1)\n\n#         # Применяем полносвязные слои с Dropout\n#         x = torch.relu(self.fc1(x))\n#         x = self.dropout(x)\n#         #x = torch.relu(self.fc2(x))\n#         #x = self.dropout(x)\n\n#         # Финальный выход (классификация)\n#         x = self.fc3(x)\n\n#         return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:24:38.355517Z","iopub.execute_input":"2024-11-24T23:24:38.356057Z","iopub.status.idle":"2024-11-24T23:24:38.365992Z","shell.execute_reply.started":"2024-11-24T23:24:38.356009Z","shell.execute_reply":"2024-11-24T23:24:38.364861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#model = tf.keras.models.load_model('/kaggle/input/best_model_v4/keras/default/1/best_model_v4.keras')\n# model = models.efficientnet_b0(pretrained=False)\n# model.classifier[1] = nn.Sequential(\n#     nn.Linear(model.classifier[1].in_features, NUM_CLASSES)\n# )\n# model.load_state_dict(torch.load('/kaggle/input/py_model_1/pytorch/default/1/best_model.pth', map_location=device))\n# model.to(device)\n# model.eval()\n\n#model = BirdSoundClassifier()\n\nmodel = models.resnet34(pretrained=False)\nnum_features = model.fc.in_features\nmodel.fc = nn.Linear(num_features, NUM_CLASSES)\nmodel.load_state_dict(torch.load('/kaggle/input/ggg/pytorch/default/1/best_model.pth', map_location=device))\nmodel.to(device)\nmodel.eval()\n\ncommon_df = pd.DataFrame()\n\nTEST_AUDIO_ROOT= \"../input/birdclef-2021/test_soundscapes/\"\nTARGET_PATH = None\nSAMPLE_SUB_PATH = \"../input/birdclef-2021/sample_submission.csv\"\n\nif not len(list(Path(TEST_AUDIO_ROOT).glob(\"*.ogg\"))):\n    TEST_AUDIO_ROOT = \"../input/birdclef-2021/train_soundscapes/\"\n    TARGET_PATH = \"../input/birdclef-2021/train_soundscape_labels.csv\"\n    SAMPLE_SUB_PATH = None\n\n\nsoundscape_files = list(listdir(TEST_AUDIO_ROOT))\ntotal_files = len(soundscape_files)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:25:31.296605Z","iopub.execute_input":"2024-11-24T23:25:31.297040Z","iopub.status.idle":"2024-11-24T23:25:32.785957Z","shell.execute_reply.started":"2024-11-24T23:25:31.297006Z","shell.execute_reply":"2024-11-24T23:25:32.784869Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Функции для обработки данных\ndef prepare_mel_spec(chunk):\n    \"\"\"Подготовка мел-спектрограммы для одного фрагмента\"\"\"\n    hop_length = int(SIGNAL_LENGTH * SAMPLE_RATE / (SPEC_SHAPE[1] - 1))\n    mel_spec = lb.feature.melspectrogram(y=chunk, \n                                              sr=SAMPLE_RATE, \n                                              n_fft=N_FFT, \n                                              hop_length=hop_length, \n                                              n_mels=SPEC_SHAPE[0], \n                                              fmin=FMIN, \n                                              fmax=FMAX)\n    mel_spec = lb.power_to_db(mel_spec, ref=np.max)\n    mel_spec -= mel_spec.min()\n    mel_spec /= mel_spec.max()\n    mel_spec = np.repeat(mel_spec[:, :, np.newaxis], 3, axis=-1)\n    return mel_spec\n\ndef process_file(soundscape_path, model, threshold=THRESHOLD):\n    \"\"\"Обработка одного файла\"\"\"\n    try:\n        y, sr = lb.load(soundscape_path, sr=SAMPLE_RATE)\n    except Exception as e:\n        print(f\"Ошибка при загрузке файла {soundscape_path}: {e}\")\n        return pd.DataFrame(columns=[\"row_id\", \"birds\"])\n\n    data = {'row_id': [], 'birds': []}\n\n    segment_length = SIGNAL_LENGTH * sr\n    sig_splits = [y[i:i + segment_length] for i in range(0, len(y), segment_length) if len(y[i:i + segment_length]) == segment_length]\n\n    # Подготовка спектрограмм для всех фрагментов\n    #mel_specs = np.array([prepare_mel_spec(chunk) for chunk in sig_splits])\n    # Предсказания для всех спектрограмм сразу\n    #predictions = model.predict(mel_specs)\n\n    mel_specs = [prepare_mel_spec(chunk) for chunk in sig_splits]\n    mel_specs = torch.stack([torch.tensor(spec, dtype=torch.float32).permute(2, 0, 1) for spec in mel_specs])  # Shape: (N, C, H, W)\n    mel_specs = mel_specs.to(device)\n    with torch.no_grad():\n        predictions = model(mel_specs)  # Shape: (N, NUM_CLASSES)\n        probabilities = F.softmax(predictions, dim=1)\n\n    seconds = 0\n    for p in probabilities:\n        seconds += 5\n        detected_species = []\n        for idx, score in enumerate(p):\n            if score.mean() > threshold:\n                detected_species.append(INV_LABEL_IDS[idx])\n        \n        prediction = \" \".join(sorted(detected_species)) if detected_species else \"nocall\"\n        row_id = f\"{soundscape_path.split(os.sep)[-1].rsplit('_', 1)[0]}_{str(seconds)}\"\n        data['row_id'].append(row_id)\n        data['birds'].append(prediction)\n\n    return pd.DataFrame(data)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:25:32.788174Z","iopub.execute_input":"2024-11-24T23:25:32.788614Z","iopub.status.idle":"2024-11-24T23:25:32.801920Z","shell.execute_reply.started":"2024-11-24T23:25:32.788567Z","shell.execute_reply":"2024-11-24T23:25:32.800626Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Обработка всех файлов\niteration = 1\nfor soundscape_path in soundscape_files:\n    soundscape_path = TEST_AUDIO_ROOT + soundscape_path\n    result_df = process_file(soundscape_path, model)\n    common_df = pd.concat([common_df, result_df])\n    \n    print(f\"ITER NUMBER {iteration} OF {total_files}\")\n    iteration += 1","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:25:32.803249Z","iopub.execute_input":"2024-11-24T23:25:32.803584Z","iopub.status.idle":"2024-11-24T23:27:06.905132Z","shell.execute_reply.started":"2024-11-24T23:25:32.803551Z","shell.execute_reply":"2024-11-24T23:27:06.903712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if SAMPLE_SUB_PATH:\n    sample_sub = pd.read_csv(SAMPLE_SUB_PATH, usecols=[\"row_id\"])\n    common_df = sample_sub.merge(common_df, on=\"row_id\", how=\"left\")\n    common_df[\"birds\"] = common_df[\"birds\"].fillna(\"nocall\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:27:06.907474Z","iopub.execute_input":"2024-11-24T23:27:06.908043Z","iopub.status.idle":"2024-11-24T23:27:06.914275Z","shell.execute_reply.started":"2024-11-24T23:27:06.908006Z","shell.execute_reply":"2024-11-24T23:27:06.913029Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Запись результата в файл\ncommon_df.to_csv('submission.csv', index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:27:06.915894Z","iopub.execute_input":"2024-11-24T23:27:06.916345Z","iopub.status.idle":"2024-11-24T23:27:06.943816Z","shell.execute_reply.started":"2024-11-24T23:27:06.916295Z","shell.execute_reply":"2024-11-24T23:27:06.942578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_metrics(s_true, s_pred):\n    s_true = set(s_true.split())\n    s_pred = set(s_pred.split())\n    n, n_true, n_pred = len(s_true.intersection(s_pred)), len(s_true), len(s_pred)\n    \n    prec = n/n_pred\n    rec = n/n_true\n    f1 = 2*prec*rec/(prec + rec) if prec + rec else 0\n    \n    return {\"f1\": f1, \"prec\": prec, \"rec\": rec, \"n_true\": n_true, \"n_pred\": n_pred, \"n\": n}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:27:06.945195Z","iopub.execute_input":"2024-11-24T23:27:06.945521Z","iopub.status.idle":"2024-11-24T23:27:06.953309Z","shell.execute_reply.started":"2024-11-24T23:27:06.945485Z","shell.execute_reply":"2024-11-24T23:27:06.951835Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if TARGET_PATH:\n    sub_target = pd.read_csv(TARGET_PATH)\n    sub_target = sub_target.merge(common_df, how=\"left\", on=\"row_id\")\n    \n    print(sub_target[\"birds_x\"].notnull().sum(), sub_target[\"birds_x\"].notnull().sum())\n    assert sub_target[\"birds_x\"].notnull().all()\n    assert sub_target[\"birds_y\"].notnull().all()\n    \n    df_metrics = pd.DataFrame([get_metrics(s_true, s_pred) for s_true, s_pred in zip(sub_target.birds_x, sub_target.birds_y)])\n    \n    print(df_metrics.mean())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:27:06.954880Z","iopub.execute_input":"2024-11-24T23:27:06.955240Z","iopub.status.idle":"2024-11-24T23:27:07.002981Z","shell.execute_reply.started":"2024-11-24T23:27:06.955208Z","shell.execute_reply":"2024-11-24T23:27:07.001642Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_target.head(15)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:27:07.006080Z","iopub.execute_input":"2024-11-24T23:27:07.006552Z","iopub.status.idle":"2024-11-24T23:27:07.026096Z","shell.execute_reply.started":"2024-11-24T23:27:07.006503Z","shell.execute_reply":"2024-11-24T23:27:07.024942Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_target[sub_target.birds_y != \"nocall\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:27:07.027369Z","iopub.execute_input":"2024-11-24T23:27:07.027697Z","iopub.status.idle":"2024-11-24T23:27:07.044027Z","shell.execute_reply.started":"2024-11-24T23:27:07.027667Z","shell.execute_reply":"2024-11-24T23:27:07.042870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sub_target[sub_target.birds_x != \"nocall\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-24T23:27:07.045420Z","iopub.execute_input":"2024-11-24T23:27:07.046481Z","iopub.status.idle":"2024-11-24T23:27:07.066227Z","shell.execute_reply.started":"2024-11-24T23:27:07.046441Z","shell.execute_reply":"2024-11-24T23:27:07.065050Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}