{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":21669,"databundleVersionId":1692278,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install efficientnet_pytorch -q\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport numpy as np\nimport random\nimport copy\nimport warnings\nimport librosa\nimport csv\nimport os\nimport pandas as pd\n\nfrom skimage.transform import resize\nfrom skimage.filters import gaussian\nfrom skimage.color import rgb2gray\nfrom skimage import exposure, util\nfrom efficientnet_pytorch import EfficientNet\nfrom torch.utils.data import Dataset, DataLoader\nfrom tqdm import tqdm\nfrom sklearn.model_selection import KFold","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-14T21:47:42.312264Z","iopub.execute_input":"2025-12-14T21:47:42.312504Z","iopub.status.idle":"2025-12-14T21:49:12.081418Z","shell.execute_reply.started":"2025-12-14T21:47:42.312478Z","shell.execute_reply":"2025-12-14T21:49:12.080556Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"warnings.filterwarnings('ignore')\ndevice = 'cuda' if torch.cuda.is_available() else 'cpu'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T21:49:12.082261Z","iopub.execute_input":"2025-12-14T21:49:12.082647Z","iopub.status.idle":"2025-12-14T21:49:12.136203Z","shell.execute_reply.started":"2025-12-14T21:49:12.082624Z","shell.execute_reply":"2025-12-14T21:49:12.135581Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_labels = 24\nlearning_rate = 2e-4\nepochs = 20\nn_folds = 5\nsr = 48000\naudio_length = 10 * sr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T21:49:12.137909Z","iopub.execute_input":"2025-12-14T21:49:12.138120Z","iopub.status.idle":"2025-12-14T21:49:12.154014Z","shell.execute_reply.started":"2025-12-14T21:49:12.138102Z","shell.execute_reply":"2025-12-14T21:49:12.153480Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def add_channels(img):\n    return np.stack((img, img, img))\n\ndef horizontal_flip(img):\n    return add_channels(img[:, ::-1])\n\ndef vertical_flip(img):\n    return add_channels(img[::-1, :])\n\ndef add_noise(img):\n    return add_channels(util.random_noise(img))\n\ndef contrast_stretching(img):\n    return add_channels(exposure.rescale_intensity(img))\n\ndef random_gaussian(img):\n    return add_channels(gaussian(img))\n\ndef random_gamma(img):\n    return add_channels(exposure.adjust_gamma(img))\n\ndef gray_scale(img):\n    return add_channels(rgb2gray(img))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T21:49:12.155098Z","iopub.execute_input":"2025-12-14T21:49:12.155384Z","iopub.status.idle":"2025-12-14T21:49:12.171386Z","shell.execute_reply.started":"2025-12-14T21:49:12.155360Z","shell.execute_reply":"2025-12-14T21:49:12.170832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def spec_to_image(spec):\n    spec = resize(spec, (224, 400))\n    \n    eps = 1e-6\n    mean = spec.mean()\n    std = spec.std()\n    spec_norm = (spec - mean) / (std + eps)\n    \n    spec_min = spec_norm.min()\n    spec_max = spec_norm.max()\n    spec_scaled = 255 * (spec_norm - spec_min) / (spec_max - spec_min)\n    spec_scaled = spec_scaled.astype(np.uint8)\n    \n    return spec_scaled","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T21:49:12.171945Z","iopub.execute_input":"2025-12-14T21:49:12.172132Z","iopub.status.idle":"2025-12-14T21:49:12.188680Z","shell.execute_reply.started":"2025-12-14T21:49:12.172110Z","shell.execute_reply":"2025-12-14T21:49:12.188168Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AudioDataset(Dataset):\n    def __init__(self, X, y, audio_data, is_train=True):\n        self.data = []\n        self.labels = []\n        self.is_train = is_train\n        \n        self.augmentations = [\n            add_noise, contrast_stretching, random_gaussian, \n            random_gamma, vertical_flip, horizontal_flip\n        ]\n        \n        for i in range(len(X)):\n            recording_id = X[i]\n            label = y[i]\n            self.data.append(audio_data[recording_id])\n            self.labels.append(label)\n    \n    def __len__(self):\n        return len(self.data)\n    \n    def __getitem__(self, idx):\n        img = self.data[idx]\n        \n        if self.is_train:\n            aug = random.choice(self.augmentations)\n            img = aug(img)\n        else:\n            img = add_channels(img)\n        \n        img_tensor = torch.FloatTensor(img)\n        label = self.labels[idx]\n        \n        return img_tensor, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T21:49:12.189325Z","iopub.execute_input":"2025-12-14T21:49:12.189572Z","iopub.status.idle":"2025-12-14T21:49:12.205401Z","shell.execute_reply.started":"2025-12-14T21:49:12.189550Z","shell.execute_reply":"2025-12-14T21:49:12.204701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_model(model, train_loader, valid_loader, epochs, optimizer, scheduler):\n    best_acc = 0\n    best_weights = None\n    criterion = nn.CrossEntropyLoss()\n    \n    for epoch in range(1, epochs + 1):\n        model.train()\n        train_loss = []\n        \n        for x, y in train_loader:\n            x = x.to(device, dtype=torch.float32)\n            y = y.to(device, dtype=torch.long)\n            \n            optimizer.zero_grad()\n            y_hat = model(x)\n            loss = criterion(y_hat, y)\n            loss.backward()\n            train_loss.append(loss.item())\n            optimizer.step()\n        \n        model.eval()\n        val_loss = []\n        all_y = []\n        all_yhat = []\n        \n        with torch.no_grad():\n            for x, y in valid_loader:\n                x = x.to(device, dtype=torch.float32)\n                y = y.to(device, dtype=torch.long)\n                \n                y_hat = model(x)\n                loss = criterion(y_hat, y)\n                val_loss.append(loss.item())\n                \n                all_y.append(y.cpu().numpy())\n                all_yhat.append(y_hat.cpu().numpy())\n        \n        all_y = np.concatenate(all_y)\n        all_yhat = np.concatenate(all_yhat)\n        accuracy = np.mean(all_yhat.argmax(axis=1) == all_y)\n        \n        scheduler.step(np.mean(val_loss))\n        \n        if accuracy > best_acc:\n            best_acc = accuracy\n            best_weights = copy.deepcopy(model.state_dict())\n        \n        print(f\"Epoch {epoch}: train_loss={np.mean(train_loss):.4f}, \"\n              f\"val_loss={np.mean(val_loss):.4f}, accuracy={accuracy:.4f}\")\n    \n    model.load_state_dict(best_weights)\n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T21:49:12.206301Z","iopub.execute_input":"2025-12-14T21:49:12.206568Z","iopub.status.idle":"2025-12-14T21:49:12.225617Z","shell.execute_reply.started":"2025-12-14T21:49:12.206546Z","shell.execute_reply":"2025-12-14T21:49:12.224893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_model():\n    model = EfficientNet.from_pretrained('efficientnet-b0', num_classes=num_labels)\n    return model.to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T21:49:12.227466Z","iopub.execute_input":"2025-12-14T21:49:12.227665Z","iopub.status.idle":"2025-12-14T21:49:12.243896Z","shell.execute_reply.started":"2025-12-14T21:49:12.227648Z","shell.execute_reply":"2025-12-14T21:49:12.243258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data = pd.read_csv(\"../input/rfcx-species-audio-detection/train_tp.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T21:49:12.244641Z","iopub.execute_input":"2025-12-14T21:49:12.244928Z","iopub.status.idle":"2025-12-14T21:49:12.274919Z","shell.execute_reply.started":"2025-12-14T21:49:12.244905Z","shell.execute_reply":"2025-12-14T21:49:12.274447Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fmin = int(data['f_min'].min() * 0.9)\nfmax = int(data['f_max'].max() * 1.1)\nprint(f\"Частотный диапазон: {fmin} - {fmax}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T21:49:12.275563Z","iopub.execute_input":"2025-12-14T21:49:12.275776Z","iopub.status.idle":"2025-12-14T21:49:12.286934Z","shell.execute_reply.started":"2025-12-14T21:49:12.275760Z","shell.execute_reply":"2025-12-14T21:49:12.286371Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"audio_data = {}\ndata_list = []\nlabel_list = []\n\nfor i in tqdm(range(len(data))):\n    recording_id = data.recording_id.values[i]\n    species_id = int(data.species_id.values[i])\n    \n    data_list.append(recording_id)\n    label_list.append(species_id)\n    \n    wav, _ = librosa.load(f'../input/rfcx-species-audio-detection/train/{recording_id}.flac', sr=sr)\n    \n    t_min = data.t_min.values[i] * sr\n    t_max = data.t_max.values[i] * sr\n    center = (t_min + t_max) / 2\n    start = max(0, center - audio_length / 2)\n    end = min(len(wav), start + audio_length)\n    \n    if end == len(wav):\n        start = max(0, end - audio_length)\n    \n    audio_slice = wav[int(start):int(end)]\n    \n    spec = librosa.feature.melspectrogram(y=audio_slice, sr=sr, fmin=fmin, fmax=fmax)\n    spec_db = librosa.power_to_db(spec, top_db=80)\n    \n    img = spec_to_image(spec_db)\n    audio_data[recording_id] = img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T21:49:12.287690Z","iopub.execute_input":"2025-12-14T21:49:12.288135Z","iopub.status.idle":"2025-12-14T21:51:46.414132Z","shell.execute_reply.started":"2025-12-14T21:49:12.288074Z","shell.execute_reply":"2025-12-14T21:51:46.413406Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"skf = KFold(n_splits=n_folds, shuffle=True, random_state=42)\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(data_list, label_list)):\n    print(f\"\\nFold {fold + 1}/{n_folds}\")\n    \n    X_train = [data_list[i] for i in train_idx]\n    y_train = [label_list[i] for i in train_idx]\n    X_val = [data_list[i] for i in val_idx]\n    y_val = [label_list[i] for i in val_idx]\n    \n    train_dataset = AudioDataset(X_train, y_train, audio_data, is_train=True)\n    val_dataset = AudioDataset(X_val, y_val, audio_data, is_train=False)\n    \n    train_loader = DataLoader(train_dataset, batch_size=8, shuffle=True)\n    val_loader = DataLoader(val_dataset, batch_size=8, shuffle=False)\n    \n    model = get_model()\n    optimizer = optim.Adam(model.parameters(), lr=learning_rate)\n    scheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, 'min', patience=3)\n    \n    model = train_model(model, train_loader, val_loader, epochs, optimizer, scheduler)\n    \n    torch.save(model.state_dict(), f\"./model{fold}.pt\")\n    print(f\"Модель сохранена model{fold}.pt\")\n    \n    del train_dataset, val_dataset, train_loader, val_loader, model\n    torch.cuda.empty_cache()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T21:51:46.415004Z","iopub.execute_input":"2025-12-14T21:51:46.415466Z","iopub.status.idle":"2025-12-14T22:09:14.513382Z","shell.execute_reply.started":"2025-12-14T21:51:46.415446Z","shell.execute_reply":"2025-12-14T22:09:14.512604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_test_file(filename):\n    wav, _ = librosa.load(f'../input/rfcx-species-audio-detection/test/{filename}', sr=sr)\n    \n    segments = []\n    num_segments = int(np.ceil(len(wav) / audio_length))\n    \n    for i in range(num_segments):\n        start = i * audio_length\n        end = min((i + 1) * audio_length, len(wav))\n        \n        if end - start < audio_length:\n            start = max(0, len(wav) - audio_length)\n            end = len(wav)\n        \n        audio_slice = wav[int(start):int(end)]\n        \n        spec = librosa.feature.melspectrogram(y=audio_slice, sr=sr, fmin=fmin, fmax=fmax)\n        spec_db = librosa.power_to_db(spec, top_db=80)\n        \n        img = spec_to_image(spec_db)\n        img_3ch = add_channels(img)\n        \n        segments.append(img_3ch)\n    \n    return np.array(segments)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T22:09:14.514511Z","iopub.execute_input":"2025-12-14T22:09:14.514971Z","iopub.status.idle":"2025-12-14T22:09:14.520646Z","shell.execute_reply.started":"2025-12-14T22:09:14.514950Z","shell.execute_reply":"2025-12-14T22:09:14.520005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def create_submission():\n    models = []\n    for i in range(n_folds):\n        model = get_model()\n        model.load_state_dict(torch.load(f'./model{i}.pt'))\n        model.eval()\n        models.append(model)\n    \n    test_files = os.listdir('../input/rfcx-species-audio-detection/test/')\n    \n    with open('submission.csv', 'w', newline='') as f:\n        writer = csv.writer(f)\n        \n        header = ['recording_id'] + [f's{i}' for i in range(num_labels)]\n        writer.writerow(header)\n        \n        for filename in tqdm(test_files):\n            file_id = filename.split('.')[0]\n            \n            segments = load_test_file(filename)\n            \n            all_predictions = []\n            \n            for segment in segments:\n                segment_tensor = torch.FloatTensor(segment).unsqueeze(0).to(device)\n                \n                model_outputs = []\n                for model in models:\n                    with torch.no_grad():\n                        output = model(segment_tensor)\n                        model_outputs.append(output.cpu())\n                \n                avg_output = torch.mean(torch.stack(model_outputs), dim=0)\n                all_predictions.append(avg_output.numpy())\n            \n            final_prediction = np.mean(all_predictions, axis=0)[0]\n            \n            row = [file_id] + list(final_prediction)\n            writer.writerow(row)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T22:09:14.521545Z","iopub.execute_input":"2025-12-14T22:09:14.521968Z","iopub.status.idle":"2025-12-14T22:09:14.540576Z","shell.execute_reply.started":"2025-12-14T22:09:14.521939Z","shell.execute_reply":"2025-12-14T22:09:14.540000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"create_submission()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-14T22:09:14.541125Z","iopub.execute_input":"2025-12-14T22:09:14.541353Z","iopub.status.idle":"2025-12-14T22:30:33.100216Z","shell.execute_reply.started":"2025-12-14T22:09:14.541337Z","shell.execute_reply":"2025-12-14T22:30:33.099541Z"}},"outputs":[],"execution_count":null}]}