{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":25954,"databundleVersionId":2091745},{"sourceType":"datasetVersion","sourceId":1493925,"datasetId":877193,"databundleVersionId":1527916},{"sourceType":"modelInstanceVersion","sourceId":774673,"databundleVersionId":15938243,"modelInstanceId":591564,"modelId":603853}],"dockerImageVersionId":31260,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Подключение библиотек","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport cv2\nfrom pathlib import Path\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport tensorflow as tf\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\nimport shutil","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T15:20:42.159794Z","iopub.execute_input":"2026-03-06T15:20:42.160034Z","iopub.status.idle":"2026-03-06T15:20:57.868556Z","shell.execute_reply.started":"2026-03-06T15:20:42.160012Z","shell.execute_reply":"2026-03-06T15:20:57.867714Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Задание констант","metadata":{}},{"cell_type":"code","source":"# Настройки\nconfig = {\n    'SR': 32000,\n    'DURATION': 5,\n    'N_MELS': 128,\n    'IMG_SIZE': 128,\n    'BATCH_SIZE': 32,\n    'EPOCHS': 20,\n    'SEED': 42\n}\n\n# Пути\nDIR_SHORT = '/kaggle/input/birdclef-2021/train_short_audio/'\nDIR_LONG = '/kaggle/input/birdclef-2021/train_soundscapes/'\nCSV_SHORT = '/kaggle/input/birdclef-2021/train_metadata.csv'\nCSV_LONG = '/kaggle/input/birdclef-2021/train_soundscape_labels.csv'\n\n# Папка для записи картинок\nOUTPUT_DIR = Path('/kaggle/working/train_images/')\n\nnp.random.seed(config['SEED'])\ntf.random.set_seed(config['SEED'])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T15:20:57.870012Z","iopub.execute_input":"2026-03-06T15:20:57.870465Z","iopub.status.idle":"2026-03-06T15:20:57.875296Z","shell.execute_reply.started":"2026-03-06T15:20:57.870441Z","shell.execute_reply":"2026-03-06T15:20:57.874618Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Функции для предобработки данных","metadata":{}},{"cell_type":"code","source":"def create_spectrograms(df, output_dir, config):\n    if output_dir.exists():\n        print(f\"Directory {output_dir} already exists. Skipping generation.\")\n        return\n        \n    print(f\"Creating spectrograms in {output_dir}...\")\n    \n    for _, row in tqdm(df.iterrows(), total=len(df)):\n        label = row['primary_label']\n        filename = row['filename']\n        \n        # Создаем папку для класса, если ее нет\n        class_dir = output_dir / label\n        class_dir.mkdir(parents=True, exist_ok=True)\n        \n        file_path = os.path.join(DIR_SHORT, label, filename)\n\n        try:\n            audio, sr = librosa.load(file_path, sr=config['SR'])\n            \n            # Нарезаем на чанки по 5 секунд\n            step = config['SR'] * config['DURATION']\n            for i, start in enumerate(range(0, len(audio), step)):\n                chunk = audio[start:start+step]\n                \n                # Игнорируем последний кусок, если он короче\n                if len(chunk) < step:\n                    continue\n                \n                # Создаем мел-спектрограмму\n                melspec = librosa.feature.melspectrogram(\n                    y=chunk, sr=config['SR'], n_mels=config['N_MELS'], fmin=300, fmax=16000\n                )\n                melspec = librosa.power_to_db(melspec, ref=np.max)\n                \n                # Нормализация\n                min_v, max_v = melspec.min(), melspec.max()\n                if max_v - min_v > 0:\n                    img = (melspec - min_v) / (max_v - min_v)\n                else:\n                    img = np.zeros_like(melspec)\n                    \n                img_uint8 = (img * 255).astype(np.uint8)\n                \n                output_filename = f\"{Path(filename).stem}_chunk{i}.png\"\n                cv2.imwrite(str(class_dir / output_filename), img_uint8)\n\n        except Exception as e:\n            print(f\"\\nError processing {file_path}: {e}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T15:20:57.876306Z","iopub.execute_input":"2026-03-06T15:20:57.877003Z","iopub.status.idle":"2026-03-06T15:20:57.896812Z","shell.execute_reply.started":"2026-03-06T15:20:57.876979Z","shell.execute_reply":"2026-03-06T15:20:57.896102Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Препроцессинг данных для обучения","metadata":{}},{"cell_type":"code","source":"# # --- Загрузка и запуск препроцессинга ---\n# df_short = pd.read_csv(CSV_SHORT)\n# create_spectrograms(df_short, OUTPUT_DIR, config)\n\n# Использовался для генерации спектрограмм для обучения","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T15:20:57.897780Z","iopub.execute_input":"2026-03-06T15:20:57.898168Z","iopub.status.idle":"2026-03-06T15:20:57.912242Z","shell.execute_reply.started":"2026-03-06T15:20:57.898137Z","shell.execute_reply":"2026-03-06T15:20:57.911540Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Подготовка данных для обучения","metadata":{}},{"cell_type":"code","source":"# # Тренировочный датасет\n# train_ds = tf.keras.utils.image_dataset_from_directory(\n#     OUTPUT_DIR,\n#     labels='inferred',\n#     label_mode='categorical',\n#     validation_split=0.15,\n#     subset='training',\n#     seed=config['SEED'],\n#     image_size=(config['IMG_SIZE'], config['IMG_SIZE']),\n#     batch_size=config['BATCH_SIZE'],\n#     color_mode='grayscale'\n# )\n\n# # Валидационный датасет\n# val_ds = tf.keras.utils.image_dataset_from_directory(\n#     OUTPUT_DIR,\n#     labels='inferred',\n#     label_mode='categorical',\n#     validation_split=0.15,\n#     subset='validation',\n#     seed=config['SEED'],\n#     image_size=(config['IMG_SIZE'], config['IMG_SIZE']),\n#     batch_size=config['BATCH_SIZE'],\n#     color_mode='grayscale'\n# )\n\n# # Получаем имена классов, которые утилита нашла\n# CLASS_NAMES = train_ds.class_names\n# NUM_CLASSES = len(CLASS_NAMES)\n# print(f\"Found {NUM_CLASSES} classes.\")\n\n# # Оптимизация загрузки\n# AUTOTUNE = tf.data.AUTOTUNE\n# train_ds = train_ds.prefetch(buffer_size=AUTOTUNE)\n# val_ds = val_ds.prefetch(buffer_size=AUTOTUNE)\n\n# Использовался для обучения","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T15:20:57.913811Z","iopub.execute_input":"2026-03-06T15:20:57.914409Z","iopub.status.idle":"2026-03-06T15:20:57.924004Z","shell.execute_reply.started":"2026-03-06T15:20:57.914387Z","shell.execute_reply":"2026-03-06T15:20:57.923227Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Подготовка метаданных для инференса","metadata":{}},{"cell_type":"code","source":"# Загружаем метаданные, чтобы получить список классов\ndf = pd.read_csv(CSV_SHORT)\n# Получаем уникальные метки и сортируем их\nCLASS_NAMES = sorted(df['primary_label'].unique())\n\nprint(f\"Total classes recovered: {len(CLASS_NAMES)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T15:20:57.924791Z","iopub.execute_input":"2026-03-06T15:20:57.925105Z","iopub.status.idle":"2026-03-06T15:20:58.316972Z","shell.execute_reply.started":"2026-03-06T15:20:57.925083Z","shell.execute_reply":"2026-03-06T15:20:58.316185Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Создание модели для обучения","metadata":{}},{"cell_type":"code","source":"# WEIGHTS_PATH = '/kaggle/input/datasets/aeryss/keras-pretrained-models/EfficientNetB0_NoTop_ImageNet.h5' \n\n# def get_model(num_classes):\n#     base_model = EfficientNetB0(\n#         include_top=False, \n#         weights=None, \n#         # Вход теперь (128, 128, 3), мы это сделаем в первом слое\n#         input_shape=(config['IMG_SIZE'], config['IMG_SIZE'], 3) \n#     )\n#     base_model.load_weights(WEIGHTS_PATH)\n    \n#     base_model.trainable = True\n\n#     model = Sequential([\n#         # Добавляем слой для преобразования 1 канала в 3\n#         Input(shape=(config['IMG_SIZE'], config['IMG_SIZE'], 1)),\n#         Conv2D(3, (1, 1), padding='same'), # 1x1 свертка для дублирования канала\n#         base_model,\n#         GlobalAveragePooling2D(),\n#         BatchNormalization(),\n#         Dropout(0.4),\n#         Dense(num_classes, activation='softmax')\n#     ])\n    \n#     model.compile(\n#         optimizer=Adam(learning_rate=1e-3),\n#         loss='categorical_crossentropy',\n#         metrics=['accuracy']\n#     )\n#     return model\n\n# model = get_model(NUM_CLASSES)\n# model.summary()\n\n# Создание","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T15:20:58.317994Z","iopub.execute_input":"2026-03-06T15:20:58.318242Z","iopub.status.idle":"2026-03-06T15:20:58.322835Z","shell.execute_reply.started":"2026-03-06T15:20:58.318219Z","shell.execute_reply":"2026-03-06T15:20:58.322257Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Создание модели для инференса","metadata":{}},{"cell_type":"code","source":"MODEL_PATH = '/kaggle/input/models/shmlvssh/birds-shumilov-av-2304-efficientnetb0-trained/keras/default/1/bird_model_preprocessed.keras'\n\ntry:\n    model = tf.keras.models.load_model(MODEL_PATH)\n    model.summary()\nexcept Exception as e:\n    print(f\"Loading model error: {e}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T15:20:58.323685Z","iopub.execute_input":"2026-03-06T15:20:58.323912Z","iopub.status.idle":"2026-03-06T15:21:03.587715Z","shell.execute_reply.started":"2026-03-06T15:20:58.323865Z","shell.execute_reply":"2026-03-06T15:21:03.587124Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Обучение","metadata":{}},{"cell_type":"code","source":"# callbacks = [\n#     ModelCheckpoint('bird_model_preprocessed.keras', monitor='val_accuracy', save_best_only=True),\n#     ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=2),\n#     EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\n# ]\n\n# history = model.fit(\n#     train_ds,\n#     validation_data=val_ds,\n#     epochs=config['EPOCHS'],\n#     callbacks=callbacks\n# )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T15:21:03.588514Z","iopub.execute_input":"2026-03-06T15:21:03.588893Z","iopub.status.idle":"2026-03-06T15:21:03.592145Z","shell.execute_reply.started":"2026-03-06T15:21:03.588859Z","shell.execute_reply":"2026-03-06T15:21:03.591580Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Очистка памяти","metadata":{}},{"cell_type":"code","source":"if OUTPUT_DIR.exists():\n    shutil.rmtree(OUTPUT_DIR)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T15:21:03.592980Z","iopub.execute_input":"2026-03-06T15:21:03.593263Z","iopub.status.idle":"2026-03-06T15:21:03.606433Z","shell.execute_reply.started":"2026-03-06T15:21:03.593243Z","shell.execute_reply":"2026-03-06T15:21:03.605898Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Предсказание","metadata":{}},{"cell_type":"code","source":"TEST_DIR = '/kaggle/input/birdclef-2021/test_soundscapes'\n\ndef predict_submission(model, labels_list):\n    # Проверяем файлы\n\n    using_dir = TEST_DIR\n   \n    files = [f for f in os.listdir(using_dir) if f.endswith('.ogg')]\n\n    if not files:\n        using_dir = DIR_LONG\n        files = [f for f in os.listdir(using_dir) if f.endswith('.ogg')]\n    \n    submission_data = []\n    \n    if not files:\n        print(\"No test files found. Creating dummy submission.\")\n        # Создаем заглушку, чтобы Save Version не падал\n        df_sub = pd.DataFrame(columns=['row_id', 'birds'])\n        df_sub.to_csv('submission.csv', index=False)\n        return\n        \n    print(f\"Processing {len(files)} files...\")\n    \n    for filename in tqdm(files, desc=\"Predicting\"):\n        path = os.path.join(using_dir, filename)\n        parts = filename.split('_')\n        audio_id = parts[0]\n        site = parts[1] if len(parts) > 1 else 'SSW'\n        \n        # Грузим ВЕСЬ файл\n        y_full, sr = librosa.load(path, sr=config['SR'])\n        \n        step = config['SR'] * 5\n        \n        batch_imgs = []\n        seconds_list = []\n        \n        # Нарезаем файл\n        for i in range(0, len(y_full), step):\n            chunk = y_full[i:i+step]\n            if len(chunk) < step: \n                break \n            \n            melspec = librosa.feature.melspectrogram(\n                y=chunk, sr=config['SR'], n_mels=config['N_MELS'], fmin=300, fmax=16000\n            )\n            melspec = librosa.power_to_db(melspec, ref=np.max)\n            \n            min_v, max_v = melspec.min(), melspec.max()\n            if max_v - min_v > 0:\n                img = (melspec - min_v) / (max_v - min_v)\n            else:\n                img = np.zeros_like(melspec)\n            \n            img = img * 255.0\n            \n            img = cv2.resize(img, (config['IMG_SIZE'], config['IMG_SIZE']))\n            img = np.expand_dims(img, axis=-1)\n            \n            batch_imgs.append(img)\n            seconds_list.append(int((i / config['SR']) + 5))\n            \n        if not batch_imgs:\n            continue\n            \n        # Предикт батчем (сразу весь файл)\n        batch_imgs = np.array(batch_imgs)\n        preds = model.predict(batch_imgs, verbose=0)\n        \n        # Обработка результатов\n        for pred, seconds in zip(preds, seconds_list):\n            idx = np.argmax(pred)\n            prob = np.max(pred)\n            \n            if prob > 0.95: # Порог уверенности\n                label = labels_list[idx]\n            else:\n                label = \"nocall\"\n                \n            row_id = f\"{audio_id}_{site}_{seconds}\"\n            submission_data.append([row_id, label])\n            \n    df_sub = pd.DataFrame(submission_data, columns=['row_id', 'birds'])\n    df_sub.to_csv('submission.csv', index=False)\n    print(\"Saved submission.csv\", df_sub.shape)\n\npredict_submission(model, CLASS_NAMES)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T15:27:01.088712Z","iopub.execute_input":"2026-03-06T15:27:01.089304Z","iopub.status.idle":"2026-03-06T15:27:41.619794Z","shell.execute_reply.started":"2026-03-06T15:27:01.089276Z","shell.execute_reply":"2026-03-06T15:27:41.618941Z"}},"outputs":[],"execution_count":null}]}