{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21669,"databundleVersionId":1692278,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport random\nimport copy\nimport csv\n\nimport numpy as np\nimport pandas as pd\nimport librosa\n\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\n\nfrom torchvision.models import resnet50\n\nfrom skimage.transform import resize\nfrom skimage import exposure, util\n\nfrom sklearn.model_selection import KFold\n\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\nNUM_CLASSES = 24\nSR = 48000\nCLIP_SECONDS = 10\nCLIP_SAMPLES = CLIP_SECONDS * SR\n\nLEARNING_RATE = 1e-4\nEPOCHS = 20\nN_FOLDS = 5\n\ntrain_csv_path = \"/kaggle/input/rfcx-species-audio-detection/train_tp.csv\"\ntrain_audio_dir = \"/kaggle/input/rfcx-species-audio-detection/train\"\ntest_audio_dir = \"/kaggle/input/rfcx-species-audio-detection/test\"\nsample_sub_path = \"/kaggle/input/rfcx-species-audio-detection/sample_submission.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T21:10:36.744865Z","iopub.execute_input":"2025-12-15T21:10:36.745534Z","iopub.status.idle":"2025-12-15T21:10:36.750798Z","shell.execute_reply.started":"2025-12-15T21:10:36.745506Z","shell.execute_reply":"2025-12-15T21:10:36.750167Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(train_csv_path)\n\nf_min = train_df[\"f_min\"].min()\nf_max = train_df[\"f_max\"].max()\n\nf_min = int(f_min * 0.9)\nf_max = int(f_max * 1.1)\n\ndef spec_to_image(spec):\n    spec_resized = resize(spec, (224, 400))\n    eps = 1e-6\n    mean = spec_resized.mean()\n    std = spec_resized.std()\n    spec_norm = (spec_resized - mean) / (std + eps)\n    spec_min = spec_norm.min()\n    spec_max = spec_norm.max()\n    spec_scaled = 255 * (spec_norm - spec_min) / (spec_max - spec_min + eps)\n    spec_scaled = spec_scaled.astype(np.uint8)\n    return spec_scaled","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T21:10:38.509759Z","iopub.execute_input":"2025-12-15T21:10:38.510368Z","iopub.status.idle":"2025-12-15T21:10:38.565758Z","shell.execute_reply.started":"2025-12-15T21:10:38.510343Z","shell.execute_reply":"2025-12-15T21:10:38.565223Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class SimpleAugment:\n    def __init__(self):\n        self.transforms = [self.add_noise, self.change_contrast]\n\n    def add_noise(self, image):\n        noisy = util.random_noise(image)\n        return np.stack([noisy] * 3)\n\n    def change_contrast(self, image):\n        img = exposure.rescale_intensity(image)\n        return np.stack([img] * 3)\n\n    def apply(self, image):\n        func = random.choice(self.transforms)\n        return func(image)\n\nrecording_ids = train_df[\"recording_id\"].tolist()\nlabels = train_df[\"species_id\"].tolist()\n\ntrain_spects = {}\n\ndef process_one(index):\n    rec_id = recording_ids[index]\n    wav_path = os.path.join(train_audio_dir, rec_id + \".flac\")\n    wav, sr = librosa.load(wav_path, sr=None)\n\n    t_min = int(train_df.at[index, \"t_min\"] * sr)\n    t_max = int(train_df.at[index, \"t_max\"] * sr)\n\n    center = int((t_min + t_max) / 2)\n    start = max(center - CLIP_SAMPLES // 2, 0)\n    end = min(start + CLIP_SAMPLES, len(wav))\n    start = end - CLIP_SAMPLES if end - start < CLIP_SAMPLES else start\n\n    segment = wav[start:end]\n    mel = librosa.feature.melspectrogram(\n        y=segment,\n        sr=sr,\n        fmin=f_min,\n        fmax=f_max\n    )\n    mel_db = librosa.power_to_db(mel, top_db=80)\n    img = spec_to_image(mel_db)\n    return rec_id, img\n\nwith ThreadPoolExecutor() as executor:\n    results = list(executor.map(process_one, range(len(train_df))))\n\nfor rec_id, img in results:\n    train_spects[rec_id] = img\n\naugmenter = SimpleAugment()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T21:10:54.145755Z","iopub.execute_input":"2025-12-15T21:10:54.146028Z","iopub.status.idle":"2025-12-15T21:12:04.458295Z","shell.execute_reply.started":"2025-12-15T21:10:54.146007Z","shell.execute_reply":"2025-12-15T21:12:04.457627Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class SpectrogramDataset(Dataset):\n    def __init__(self, rec_ids, targets, mode):\n        self.rec_ids = rec_ids\n        self.targets = targets\n        self.mode = mode\n        self.storage = train_spects\n\n    def __len__(self):\n        return len(self.rec_ids)\n\n    def __getitem__(self, idx):\n        rec_id = self.rec_ids[idx]\n        label = self.targets[idx]\n        img = self.storage[rec_id]\n\n        if self.mode == \"train\":\n            img3 = augmenter.apply(img)\n        else:\n            img3 = np.stack([img] * 3)\n\n        return img3, label\n\ndef build_model():\n    model = resnet50(pretrained=True)\n    in_features = model.fc.in_features\n    model.fc = nn.Sequential(\n        nn.Dropout(0.3),\n        nn.Linear(in_features, in_features // 2),\n        nn.ReLU(inplace=True),\n        nn.Linear(in_features // 2, NUM_CLASSES)\n    )\n    return model.to(device)\n\ncriterion = nn.CrossEntropyLoss()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T21:12:23.560387Z","iopub.execute_input":"2025-12-15T21:12:23.560827Z","iopub.status.idle":"2025-12-15T21:12:23.567727Z","shell.execute_reply.started":"2025-12-15T21:12:23.560807Z","shell.execute_reply":"2025-12-15T21:12:23.567151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_one_fold(model, train_loader, valid_loader, optimizer, scheduler):\n    best_wts = copy.deepcopy(model.state_dict())\n    best_acc = 0.0\n\n    for epoch in range(1, EPOCHS + 1):\n        model.train()\n        train_losses = []\n\n        for batch_imgs, batch_labels in train_loader:\n            optimizer.zero_grad()\n            x = batch_imgs.to(device, dtype=torch.float32)\n            y = batch_labels.to(device, dtype=torch.long)\n            outputs = model(x)\n            loss = criterion(outputs, y)\n            loss.backward()\n            optimizer.step()\n            train_losses.append(loss.item())\n\n        model.eval()\n        valid_losses = []\n        all_true = []\n        all_pred = []\n\n        with torch.no_grad():\n            for batch_imgs, batch_labels in valid_loader:\n                x = batch_imgs.to(device, dtype=torch.float32)\n                y = batch_labels.to(device, dtype=torch.long)\n                outputs = model(x)\n                loss = criterion(outputs, y)\n                valid_losses.append(loss.item())\n                all_true.append(y.cpu().numpy())\n                all_pred.append(outputs.cpu().numpy())\n\n        all_true = np.concatenate(all_true)\n        all_pred = np.concatenate(all_pred)\n        acc = np.mean(all_pred.argmax(axis=1) == all_true)\n\n        train_loss = float(np.mean(train_losses))\n        valid_loss = float(np.mean(valid_losses))\n\n        print(\"epoch:\", epoch, \"train_loss:\", round(train_loss, 4),\n              \"val_loss:\", round(valid_loss, 4),\n              \"val_acc:\", round(acc, 4))\n\n        scheduler.step(valid_loss)\n        if acc > best_acc:\n            best_acc = acc\n            best_wts = copy.deepcopy(model.state_dict())\n\n    model.load_state_dict(best_wts)\n    return model\n\nkf = KFold(n_splits=N_FOLDS, shuffle=True, random_state=563)\n\nall_rec_ids = np.array(recording_ids)\nall_labels = np.array(labels)\n\nfor fold_idx, (train_idx, valid_idx) in enumerate(kf.split(all_rec_ids, all_labels)):\n    print(\"Fold\", fold_idx)\n    x_train = all_rec_ids[train_idx]\n    y_train = all_labels[train_idx]\n    x_valid = all_rec_ids[valid_idx]\n    y_valid = all_labels[valid_idx]\n\n    train_dataset = SpectrogramDataset(x_train, y_train, mode=\"train\")\n    valid_dataset = SpectrogramDataset(x_valid, y_valid, mode=\"valid\")\n\n    train_loader = DataLoader(train_dataset, batch_size=8, shuffle=True, drop_last=True)\n    valid_loader = DataLoader(valid_dataset, batch_size=8, shuffle=False, drop_last=False)\n\n    model = build_model()\n    optimizer = torch.optim.Adam(model.parameters(), lr=LEARNING_RATE)\n    scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode=\"min\", patience=3)\n\n    model = train_one_fold(model, train_loader, valid_loader, optimizer, scheduler)\n    torch.save(model.state_dict(), f\"model_fold_{fold_idx}.pth\")\n\n    del train_dataset, valid_dataset, train_loader, valid_loader, model\n    torch.cuda.empty_cache()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T21:12:29.792121Z","iopub.execute_input":"2025-12-15T21:12:29.792831Z","iopub.status.idle":"2025-12-15T21:50:54.624182Z","shell.execute_reply.started":"2025-12-15T21:12:29.792807Z","shell.execute_reply":"2025-12-15T21:50:54.623277Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_test_file(file_name):\n    path = os.path.join(test_audio_dir, file_name)\n    wav, sr = librosa.load(path, sr=None)\n    total_len = len(wav)\n    segments = int(np.ceil(total_len / CLIP_SAMPLES))\n\n    clips = []\n    for i in range(segments):\n        start = i * CLIP_SAMPLES\n        end = start + CLIP_SAMPLES\n        if end > total_len:\n            start = max(total_len - CLIP_SAMPLES, 0)\n            end = total_len\n        segment = wav[start:end]\n        if len(segment) < CLIP_SAMPLES:\n            pad = np.zeros(CLIP_SAMPLES - len(segment))\n            segment = np.concatenate([segment, pad])\n        mel = librosa.feature.melspectrogram(\n            y=segment,\n            sr=sr,\n            fmin=f_min,\n            fmax=f_max\n        )\n        mel_db = librosa.power_to_db(mel, top_db=80)\n        img = spec_to_image(mel_db)\n        img3 = np.stack([img] * 3)\n        clips.append(img3)\n    return np.stack(clips)\n\nsample_sub = pd.read_csv(sample_sub_path)\ntest_files = sample_sub[\"recording_id\"].tolist()\n\nfold_models = []\nfor fold_idx in range(N_FOLDS):\n    m = build_model()\n    m.load_state_dict(torch.load(f\"model_fold_{fold_idx}.pth\", map_location=device))\n    m.eval()\n    fold_models.append(m)\n\npred_rows = []\n\nfor rec_id in test_files:\n    file_name = rec_id + \".flac\"\n    batch = load_test_file(file_name)\n    batch_tensor = torch.tensor(batch, dtype=torch.float32).to(device)\n\n    with torch.no_grad():\n        fold_probs = []\n        for m in fold_models:\n            outputs = m(batch_tensor)\n            probs = torch.softmax(outputs, dim=1).cpu().numpy()\n            fold_probs.append(probs)\n        fold_probs = np.mean(fold_probs, axis=0)\n\n    mean_probs = fold_probs.mean(axis=0)\n\n    row = [rec_id] + [float(mean_probs[c]) for c in range(NUM_CLASSES)]\n    pred_rows.append(row)\n\ncols = [\"recording_id\"] + [f\"s{i}\" for i in range(NUM_CLASSES)]\npred_df = pd.DataFrame(pred_rows, columns=cols)\npred_df.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T21:52:52.017498Z","iopub.execute_input":"2025-12-15T21:52:52.018215Z","iopub.status.idle":"2025-12-15T22:08:43.132102Z","shell.execute_reply.started":"2025-12-15T21:52:52.018163Z","shell.execute_reply":"2025-12-15T22:08:43.131258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nsub = pd.read_csv(\"submission.csv\")\nsub.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T22:09:09.866316Z","iopub.execute_input":"2025-12-15T22:09:09.867019Z","iopub.status.idle":"2025-12-15T22:09:09.913774Z","shell.execute_reply.started":"2025-12-15T22:09:09.866997Z","shell.execute_reply":"2025-12-15T22:09:09.912985Z"}},"outputs":[],"execution_count":null}]}