{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\ni = 0\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        i += 1\n        if i == 30:\n            brack\n        print(os.path.join(dirname, filename))\n        pass\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-10T07:20:26.570775Z","iopub.execute_input":"2023-05-10T07:20:26.571162Z","iopub.status.idle":"2023-05-10T07:20:26.634695Z","shell.execute_reply.started":"2023-05-10T07:20:26.571133Z","shell.execute_reply":"2023-05-10T07:20:26.633153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom PIL import Image\n\ndef process_audio_file(audio_path):\n    # Загружаем аудиоданные\n    audio, sr = librosa.load(audio_path, sr=None, mono=True)\n    \n    # Выбираем случайный отрезок длиной не более 5 секунд\n    audio_duration = len(audio) / sr\n    max_audio_offset = max(audio_duration - 5, 0)\n    audio_offset = np.random.uniform(0, max_audio_offset)\n    audio = audio[int(audio_offset * sr):int((audio_offset + 5) * sr)]\n    \n    # Создаем спектрограмму\n    n_fft = 2048\n    hop_length = 512\n    spectrogram = librosa.stft(audio, n_fft=n_fft, hop_length=hop_length)\n    spectrogram = librosa.amplitude_to_db(np.abs(spectrogram), ref=np.max)\n    \n    # Изменяем размер до 1024x1024\n    img = Image.fromarray(spectrogram)\n    img = img.resize((1024, 1024), resample=Image.LANCZOS)\n    spectrogram = np.array(img)\n    \n    # Нормализуем спектрограмму\n    mean = np.mean(spectrogram)\n    std = np.std(spectrogram)\n    spectrogram = (spectrogram - mean) / std\n    \n    return spectrogram\n","metadata":{"execution":{"iopub.status.busy":"2023-05-10T07:30:41.828875Z","iopub.execute_input":"2023-05-10T07:30:41.829297Z","iopub.status.idle":"2023-05-10T07:30:41.840643Z","shell.execute_reply.started":"2023-05-10T07:30:41.829263Z","shell.execute_reply":"2023-05-10T07:30:41.839361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport matplotlib.pyplot as plt\n\n# Загружаем пример аудиофайла\naudio_path = 'example_audio.wav'\n\n# Создаем спектрограмму с помощью функции process_audio_file\nspectrogram = process_audio_file(\"/kaggle/input/birdclef-2023/train_audio/yetgre1/XC346069.ogg\")\n\n# Визуализируем спектрограмму\nplt.imshow(spectrogram, cmap='gray')\nplt.axis('off')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-10T07:30:42.108169Z","iopub.execute_input":"2023-05-10T07:30:42.108817Z","iopub.status.idle":"2023-05-10T07:30:42.377731Z","shell.execute_reply.started":"2023-05-10T07:30:42.108782Z","shell.execute_reply":"2023-05-10T07:30:42.376863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport random\nfrom PIL import Image\nimport numpy as np\nfrom torch.utils.data import DataLoader, Dataset\nimport torch\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(device)\nimport torch.nn as nn\nimport torch.optim as optim\nimport torchvision.transforms as transforms\nimport torchvision.transforms.functional as F\nimport os\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom PIL import Image\nfrom pydub import AudioSegment\nimport matplotlib.pyplot as plt\nimport librosa.display\nfrom torchvision import transforms\nimport torch\nimport os\nimport glob\nfrom torchlibrosa.stft import Spectrogram\nimport librosa\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom timm.models.resnest import resnest50d\nimport torch.optim as optim\nimport math\nfrom resnest.torch import resnest50\ntorch.backends.cudnn.benchmark = True","metadata":{"execution":{"iopub.status.busy":"2023-05-10T07:30:42.379452Z","iopub.execute_input":"2023-05-10T07:30:42.379953Z","iopub.status.idle":"2023-05-10T07:30:42.390887Z","shell.execute_reply.started":"2023-05-10T07:30:42.379921Z","shell.execute_reply":"2023-05-10T07:30:42.389850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport glob\nimport numpy as np\nimport torch\nfrom torch.utils.data import Dataset\nimport librosa\nfrom torchvision import transforms\n\ntransform = transforms.Compose([\n    transforms.Lambda(lambda x: x.unsqueeze(0)),  \n])\n\nclass CustomDataset(Dataset):\n    def __init__(self, path, transform=None, sr=32000, device='cuda'):\n        self.device = device\n        self.transform = transform\n        self.sr = sr\n        self.classes = os.listdir(path)\n        self.path_to_audio = []\n        self.ds_labels = []\n        self.ds_class_to_idx = {} \n        idx = 0\n        for i in self.classes:\n            self.path_to_audio.extend(glob.glob(path + i + '/*.ogg'))\n            self.ds_labels.extend([i]*len(glob.glob(path + i + '/*.ogg')))\n            self.ds_class_to_idx[i] = idx # добавляем в словарь новый класс и присваиваем ему уникальное числовое значение\n            idx += 1\n        self.ds_labels = [self.ds_class_to_idx[label] for label in self.ds_labels]\n\n\n    def __len__(self):\n        return len(self.ds_labels)\n\n    def process_audio_file(self, audio_path, sr = 32000):\n        # Загружаем аудиоданные\n        audio, _ = librosa.load(audio_path, sr=sr, mono=True)\n\n        # Выбираем случайный отрезок длиной не более 5 секунд\n        audio_duration = len(audio) / sr\n        max_audio_offset = max(audio_duration - 5, 0)\n        audio_offset = np.random.uniform(0, max_audio_offset)\n        audio = audio[int(audio_offset * sr):int((audio_offset + 5) * sr)]\n\n        # Создаем спектрограмму\n        n_fft = 2048\n        hop_length = 512\n        spectrogram = librosa.stft(audio, n_fft=n_fft, hop_length=hop_length)\n        spectrogram = librosa.amplitude_to_db(np.abs(spectrogram), ref=np.max)\n\n        # Изменяем размер до 1024x1024\n        img = Image.fromarray(spectrogram)\n        img = img.resize((1024, 1024), resample=Image.LANCZOS)\n        spectrogram = np.array(img)\n\n        # Нормализуем спектрограмму\n        mean = np.mean(spectrogram)\n        std = np.std(spectrogram)\n        spectrogram = (spectrogram - mean) / std\n\n        return spectrogram\n\n\n\n    def __getitem__(self, idx):\n        spec = self.process_audio_file(self.path_to_audio[idx])\n        spec = torch.from_numpy(spec)\n        if self.transform is not None:\n            spec = self.transform(spec)\n        return spec, self.ds_labels[idx]","metadata":{"execution":{"iopub.status.busy":"2023-05-10T07:30:44.062518Z","iopub.execute_input":"2023-05-10T07:30:44.062876Z","iopub.status.idle":"2023-05-10T07:30:44.079292Z","shell.execute_reply.started":"2023-05-10T07:30:44.062846Z","shell.execute_reply":"2023-05-10T07:30:44.078383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size=8\n\ndataset = CustomDataset(path=\"/kaggle/input/birdclef-2023/train_audio/\", transform=transform)\nn_samples = len(dataset)\nn_val_samples = math.floor(0.2 * n_samples)\nn_train_samples = n_samples - n_val_samples\n\ntrain_dataset, val_dataset = torch.utils.data.random_split(dataset, [n_train_samples, n_val_samples])\ntrain_dataloader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True)\nval_dataloader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-10T07:30:44.081283Z","iopub.execute_input":"2023-05-10T07:30:44.081653Z","iopub.status.idle":"2023-05-10T07:30:47.336043Z","shell.execute_reply.started":"2023-05-10T07:30:44.081621Z","shell.execute_reply":"2023-05-10T07:30:47.334732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from efficientnet_pytorch import EfficientNet\n\nclass EfficientNetB7(nn.Module):\n    def init(self, num_classes=264):\n        super(EfficientNetB7, self).init()\n        self.efficientnet = EfficientNet.from_pretrained('efficientnet-b7')\n        self.num_classes = num_classes\n        \n        self.fc1 = nn.Linear(self.efficientnet._fc.in_features, 4096)\n        self.fc2 = nn.Linear(4096, 512)\n        self.fc3 = nn.Linear(512, num_classes)\n        \n    def forward(self, x):\n        # Конвертируем входной тензор в тип float\n        x = x.float()\n        \n        x = self.efficientnet(x)\n        \n        # Применяем три новых полносвязанных слоя\n        x = self.fc1(x)\n        x = nn.functional.relu(x)\n        x = self.fc2(x)\n        x = nn.functional.relu(x)\n        x = self.fc3(x)\n        \n        return x\n\n\nclass EfficientNetResNet(nn.Module):\n    def init(self, num_classes=264):\n        super(EfficientNetResNet, self).init()\n        self.resnet = models.resnet101(pretrained=True)\n        self.efficientnet = EfficientNet.from_pretrained('efficientnet-b7')\n        self.num_classes = num_classes\n        \n        # Заменяем первый сверточный слой ResNet на аналогичный слой EfficientNet\n        self.resnet.conv1 = nn.Conv2d(3, 64, kernel_size=3, stride=2, padding=1, bias=False)\n        self.efficientnet._conv_stem = self.resnet.conv1\n        \n        # Заменяем последний полносвязанный слой EfficientNet на новый полносвязанный слой\n        self.efficientnet._fc = nn.Linear(self.efficientnet._fc.in_features, 4096)\n        \n        # Добавляем два новых полносвязанных слоя\n        self.fc1 = nn.Linear(4096 + self.resnet.fc.in_features, 2048)\n        self.fc2 = nn.Linear(2048, num_classes)\n        \n    def forward(self, x):\n        # Конвертируем входной тензор в тип float\n        x = x.float()\n        \n        x_resnet = self.resnet(x)\n        x_efficientnet = self.efficientnet(x)\n        \n        # Объединяем выходы ResNet и EfficientNet\n        x = torch.cat((x_resnet, x_efficientnet), dim=1)\n        \n        # Применяем два новых полносвязанных слоя\n        x = self.fc1(x)\n        x = nn.functional.relu(x)\n        x = self.fc2(x)\n        \n        return x\n\n\nmodel = EfficientNetResNet().to(device)","metadata":{"execution":{"iopub.status.busy":"2023-05-10T07:36:15.949661Z","iopub.execute_input":"2023-05-10T07:36:15.950043Z","iopub.status.idle":"2023-05-10T07:36:15.965544Z","shell.execute_reply.started":"2023-05-10T07:36:15.950012Z","shell.execute_reply":"2023-05-10T07:36:15.964610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! pip install efficientnet_pytorch","metadata":{"execution":{"iopub.status.busy":"2023-05-10T07:32:09.740887Z","iopub.execute_input":"2023-05-10T07:32:09.741263Z","iopub.status.idle":"2023-05-10T07:32:23.267532Z","shell.execute_reply.started":"2023-05-10T07:32:09.741232Z","shell.execute_reply":"2023-05-10T07:32:23.266139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-10T07:30:57.996643Z","iopub.execute_input":"2023-05-10T07:30:57.997533Z","iopub.status.idle":"2023-05-10T07:30:58.030097Z","shell.execute_reply.started":"2023-05-10T07:30:57.997498Z","shell.execute_reply":"2023-05-10T07:30:58.028473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import logging\nlogger.handlers.clear()\n\n# Создаем логгер\nlogger = logging.getLogger('training_log')\nlogger.propagate = False # Предотвращаем повторные выводы\nlogger.setLevel(logging.DEBUG)\n\n# Создаем обработчик для записи сообщений в файл\nfile_handler = logging.FileHandler('training_log_Model_2.txt')\nfile_handler.setLevel(logging.DEBUG)\nfile_formatter = logging.Formatter('%(asctime)s - %(message)s')\nfile_handler.setFormatter(file_formatter)\n\n# Создаем обработчик для консольного вывода\nconsole_handler = logging.StreamHandler()\nconsole_handler.setLevel(logging.DEBUG)\nconsole_formatter = logging.Formatter('%(asctime)s - %(message)s')\nconsole_handler.setFormatter(console_formatter)\n\n# Добавляем обработчики к логгеру\nlogger.addHandler(file_handler)\nlogger.addHandler(console_handler)\nlogger.info(12)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-10T07:34:34.666317Z","iopub.execute_input":"2023-05-10T07:34:34.666677Z","iopub.status.idle":"2023-05-10T07:34:34.678478Z","shell.execute_reply.started":"2023-05-10T07:34:34.666648Z","shell.execute_reply":"2023-05-10T07:34:34.677587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"criterion = nn.CrossEntropyLoss()\n\noptimizer = optim.SGD(model.parameters(), lr=0.001)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-10T07:36:26.660982Z","iopub.execute_input":"2023-05-10T07:36:26.661622Z","iopub.status.idle":"2023-05-10T07:36:26.716331Z","shell.execute_reply.started":"2023-05-10T07:36:26.661588Z","shell.execute_reply":"2023-05-10T07:36:26.715017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\nbest_val_loss = float('inf')\nbest_model = None\nbest_val_acc = 0.0\nnum_epochs = 100\n\nfor epoch in range(num_epochs):\n    logger.info(f\"Epoch {epoch+1}/{num_epochs}\")\n    \n    running_loss = 0.0\n    model.train()\n    for i, (images, labels) in enumerate(train_dataloader):\n        optimizer.zero_grad()\n        images, labels = images.to(device), labels.to(device)\n        output = model(images)\n        loss = criterion(output, labels)\n        loss.backward()\n        optimizer.step()\n\n        running_loss += loss.item() * len(images)\n        if (i+1) % 3 == 0:\n            logger.info(f\"Train Batch {i+1}/{len(train_dataloader)}: Loss {loss.item():.4f}\")\n    \n    train_loss = running_loss / len(train_dataset)\n    logger.info(f\"Train Loss: {train_loss:.4f}\")\n    \n    model.eval()\n    running_val_loss = 0.0\n    y_true = []\n    y_pred = []\n\n    with torch.no_grad():\n        for images, labels in val_dataloader:\n            images, labels = images.to(device), labels.to(device)\n            output = model(images)\n            val_loss = criterion(output, labels)\n            running_val_loss += val_loss.item() * len(images)\n            pred = output.argmax(dim=1, keepdim=True)\n            y_pred += pred.cpu().numpy().tolist()\n            y_true += labels.cpu().numpy().tolist()\n\n        running_val_acc = accuracy_score(y_true, y_pred)\n        logger.info(f\"Validation Accuracy: {running_val_acc:.4f}\")\n\n    if best_val_acc < running_val_acc:\n        logger.info(f\"Save Model: {running_val_acc:.4f}\")\n\n        best_val_acc = running_val_acc\n        best_model = model.state_dict()\n        torch.save(best_model, 'model_birds_2_RAM_LIST.pth')\n    \nlogger.info(f\"Best Validation Loss: {best_val_loss:.4f} Score: {best_val_acc:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-10T07:36:29.556938Z","iopub.execute_input":"2023-05-10T07:36:29.557309Z","iopub.status.idle":"2023-05-10T07:36:30.273326Z","shell.execute_reply.started":"2023-05-10T07:36:29.557278Z","shell.execute_reply":"2023-05-10T07:36:30.272008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}