{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":8900,"databundleVersionId":862232,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport sys\nimport glob\nimport math\nimport random\nfrom time import time\nfrom pathlib import Path\n\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport soundfile as sf\nfrom sklearn.model_selection import train_test_split\n\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda.amp import autocast, GradScaler","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:46.790377Z","iopub.execute_input":"2025-11-10T15:22:46.790585Z","iopub.status.idle":"2025-11-10T15:22:52.945878Z","shell.execute_reply.started":"2025-11-10T15:22:46.790560Z","shell.execute_reply":"2025-11-10T15:22:52.945226Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"ROOT_INPUT = \"/kaggle/input\"\nSR = 32000\nDURATION = 4.0\nSAMPLES = int(SR * DURATION)\nN_MELS = 128\nN_FFT = 2048\nHOP_LENGTH = 512\nFMIN = 20\nFMAX = SR // 2\nPOWER = 2.0\n\nBATCH_SIZE = 32\nEPOCHS = 8\nLR = 1e-3\nNUM_WORKERS = 2\nSEED = 42\nDEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nPRINT_EVERY = 20","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:52.947518Z","iopub.execute_input":"2025-11-10T15:22:52.947877Z","iopub.status.idle":"2025-11-10T15:22:53.011459Z","shell.execute_reply.started":"2025-11-10T15:22:52.947857Z","shell.execute_reply":"2025-11-10T15:22:53.010801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"random.seed(SEED)\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\nif DEVICE == \"cuda\":\n    torch.cuda.manual_seed_all(SEED)\n\nprint(\"Device:\", DEVICE)\nprint(\"Searching under\", ROOT_INPUT)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.012172Z","iopub.execute_input":"2025-11-10T15:22:53.012402Z","iopub.status.idle":"2025-11-10T15:22:53.029751Z","shell.execute_reply.started":"2025-11-10T15:22:53.012375Z","shell.execute_reply":"2025-11-10T15:22:53.029076Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_csv_path = \"/kaggle/input/freesound-audio-tagging/train.csv\"\ntest_csv_path  = \"/kaggle/input/freesound-audio-tagging/test_post_competition.csv\"\nsample_sub_path = \"/kaggle/input/freesound-audio-tagging/sample_submission.csv\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.030518Z","iopub.execute_input":"2025-11-10T15:22:53.030812Z","iopub.status.idle":"2025-11-10T15:22:53.038130Z","shell.execute_reply.started":"2025-11-10T15:22:53.030794Z","shell.execute_reply":"2025-11-10T15:22:53.037491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv(train_csv_path)\ntest_df  = pd.read_csv(test_csv_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.038970Z","iopub.execute_input":"2025-11-10T15:22:53.039209Z","iopub.status.idle":"2025-11-10T15:22:53.113895Z","shell.execute_reply.started":"2025-11-10T15:22:53.039188Z","shell.execute_reply":"2025-11-10T15:22:53.113181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"unique_labels = sorted(train_df['label'].unique().tolist())\nLABELS = unique_labels\nNUM_CLASSES = len(LABELS)\nlabel_to_idx = {l: i for i, l in enumerate(LABELS)}\nprint(\"Extracted labels ({}):\".format(NUM_CLASSES), LABELS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.114691Z","iopub.execute_input":"2025-11-10T15:22:53.115013Z","iopub.status.idle":"2025-11-10T15:22:53.124875Z","shell.execute_reply.started":"2025-11-10T15:22:53.114986Z","shell.execute_reply":"2025-11-10T15:22:53.124304Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_audio_dir = \"/kaggle/input/freesound-audio-tagging/audio_train\"\ntrain_count = 9473\ntest_audio_dir = \"/kaggle/input/freesound-audio-tagging/audio_test\"\ntest_count = 9400","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.127201Z","iopub.execute_input":"2025-11-10T15:22:53.127477Z","iopub.status.idle":"2025-11-10T15:22:53.137904Z","shell.execute_reply.started":"2025-11-10T15:22:53.127459Z","shell.execute_reply":"2025-11-10T15:22:53.137221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_audio(path, sr=SR, samples=SAMPLES):\n    try:\n        data, orig_sr = sf.read(path, dtype='float32')\n    except Exception:\n        data, orig_sr = librosa.load(path, sr=None)\n    if data.ndim > 1:\n        data = np.mean(data, axis=1)\n    if orig_sr != sr:\n        data = librosa.resample(y=data, orig_sr=orig_sr, target_sr=sr)\n    if len(data) < samples:\n        pad = samples - len(data)\n        data = np.pad(data, (0, pad), mode='constant')\n    else:\n        data = data[:samples]\n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.138542Z","iopub.execute_input":"2025-11-10T15:22:53.138788Z","iopub.status.idle":"2025-11-10T15:22:53.153101Z","shell.execute_reply.started":"2025-11-10T15:22:53.138763Z","shell.execute_reply":"2025-11-10T15:22:53.152414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def wav_to_log_mel(wav, sr=SR, n_mels=N_MELS, n_fft=N_FFT, hop_length=HOP_LENGTH, fmin=FMIN, fmax=FMAX, power=POWER):\n    mel = librosa.feature.melspectrogram(\n        y=wav, sr=sr, n_fft=n_fft, hop_length=hop_length,\n        n_mels=n_mels, fmin=fmin, fmax=fmax, power=power\n    )\n    log_mel = librosa.power_to_db(mel, ref=np.max)\n    return log_mel.astype(np.float32)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.153782Z","iopub.execute_input":"2025-11-10T15:22:53.153942Z","iopub.status.idle":"2025-11-10T15:22:53.169956Z","shell.execute_reply.started":"2025-11-10T15:22:53.153929Z","shell.execute_reply":"2025-11-10T15:22:53.169334Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def label_to_multihot_str(label_str):\n    arr = np.zeros(NUM_CLASSES, dtype=np.float32)\n    if isinstance(label_str, str) and label_str.strip() != '':\n        parts = label_str.strip().split()\n        for p in parts:\n            if p in label_to_idx:\n                arr[label_to_idx[p]] = 1.0\n            else:\n                if p.lower() in {k.lower(): v for k, v in label_to_idx.items()}:\n                    # map lower-case back\n                    for k, v in label_to_idx.items():\n                        if k.lower() == p.lower():\n                            arr[v] = 1.0\n                            break\n                else:\n                    # ignore unknown\n                    pass\n    return arr","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.170806Z","iopub.execute_input":"2025-11-10T15:22:53.171064Z","iopub.status.idle":"2025-11-10T15:22:53.182241Z","shell.execute_reply.started":"2025-11-10T15:22:53.171041Z","shell.execute_reply":"2025-11-10T15:22:53.181470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class FreesoundDataset(Dataset):\n    def __init__(self, df, audio_dir, is_test=False):\n        self.df = df.reset_index(drop=True)\n        self.audio_dir = audio_dir\n        self.is_test = is_test\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        fname = row['fname']\n        path = os.path.join(self.audio_dir, fname)\n        if not os.path.exists(path):\n            found = list(Path(self.audio_dir).rglob(fname))\n            if found:\n                path = str(found[0])\n            else:\n                raise FileNotFoundError(f\"{path} not found\")\n        wav = load_audio(path)\n        feat = wav_to_log_mel(wav)  # shape (n_mels, time)\n        feat = (feat - feat.mean()) / (feat.std() + 1e-9)\n        x = torch.from_numpy(feat).unsqueeze(0)  # (1, n_mels, time)\n        if self.is_test:\n            return x.float(), fname\n        else:\n            lab = row['label']\n            y = label_to_multihot_str(lab)\n            return x.float(), torch.from_numpy(y)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.182992Z","iopub.execute_input":"2025-11-10T15:22:53.183203Z","iopub.status.idle":"2025-11-10T15:22:53.196664Z","shell.execute_reply.started":"2025-11-10T15:22:53.183182Z","shell.execute_reply":"2025-11-10T15:22:53.195840Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ConvBlock(nn.Module):\n    def __init__(self, in_ch, out_ch, pool=True):\n        super().__init__()\n        self.conv = nn.Sequential(\n            nn.Conv2d(in_ch, out_ch, kernel_size=3, padding=1, bias=False),\n            nn.BatchNorm2d(out_ch),\n            nn.ReLU(inplace=True),\n            nn.Conv2d(out_ch, out_ch, kernel_size=3, padding=1, bias=False),\n            nn.BatchNorm2d(out_ch),\n            nn.ReLU(inplace=True),\n        )\n        self.pool = nn.MaxPool2d(2) if pool else nn.Identity()\n\n    def forward(self, x):\n        x = self.conv(x)\n        x = self.pool(x)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.197357Z","iopub.execute_input":"2025-11-10T15:22:53.197574Z","iopub.status.idle":"2025-11-10T15:22:53.213657Z","shell.execute_reply.started":"2025-11-10T15:22:53.197552Z","shell.execute_reply":"2025-11-10T15:22:53.212920Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class AudioCNN(nn.Module):\n    def __init__(self, n_classes=NUM_CLASSES, in_ch=1):\n        super().__init__()\n        self.enc = nn.Sequential(\n            ConvBlock(in_ch, 16),\n            ConvBlock(16, 32),\n            ConvBlock(32, 64),\n            ConvBlock(64, 128),\n            ConvBlock(128, 256, pool=False),\n        )\n        self.global_pool = nn.AdaptiveAvgPool2d((1, 1))\n        self.fc = nn.Linear(256, n_classes)\n\n    def forward(self, x):\n        x = self.enc(x)\n        x = self.global_pool(x)\n        x = x.view(x.size(0), -1)\n        x = self.fc(x)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.214355Z","iopub.execute_input":"2025-11-10T15:22:53.214546Z","iopub.status.idle":"2025-11-10T15:22:53.226703Z","shell.execute_reply.started":"2025-11-10T15:22:53.214531Z","shell.execute_reply":"2025-11-10T15:22:53.225903Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def apk(actual, predicted, k=3):\n    if len(predicted) > k:\n        predicted = predicted[:k]\n    score = 0.0\n    hits = 0.0\n    for i, p in enumerate(predicted):\n        if p in actual and p not in predicted[:i]:\n            hits += 1.0\n            score += hits / (i + 1.0)\n    # normalize by min(len(actual), k)\n    denom = min(len(actual), k)\n    return score / denom if denom > 0 else 0.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.227441Z","iopub.execute_input":"2025-11-10T15:22:53.227646Z","iopub.status.idle":"2025-11-10T15:22:53.242187Z","shell.execute_reply.started":"2025-11-10T15:22:53.227622Z","shell.execute_reply":"2025-11-10T15:22:53.241453Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def mapk(actuals, predicteds, k=3):\n    return np.mean([apk(a, p, k) for a, p in zip(actuals, predicteds)])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.242957Z","iopub.execute_input":"2025-11-10T15:22:53.243632Z","iopub.status.idle":"2025-11-10T15:22:53.254234Z","shell.execute_reply.started":"2025-11-10T15:22:53.243614Z","shell.execute_reply":"2025-11-10T15:22:53.253504Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_one_epoch(model, loader, optimizer, scaler, epoch):\n    model.train()\n    criterion = nn.BCEWithLogitsLoss()\n    running_loss = 0.0\n    n = 0\n    t0 = time()\n    for i, (x, y) in enumerate(loader):\n        x = x.to(DEVICE)\n        y = y.to(DEVICE)\n        optimizer.zero_grad()\n        with autocast():\n            logits = model(x)\n            loss = criterion(logits, y)\n        scaler.scale(loss).backward()\n        scaler.step(optimizer)\n        scaler.update()\n        running_loss += loss.item() * x.size(0)\n        n += x.size(0)\n        if i % PRINT_EVERY == 0:\n            print(f\"Epoch {epoch} iter {i}/{len(loader)} loss {loss.item():.4f}\")\n    avg = running_loss / n\n    print(f\"Epoch {epoch} finished in {time()-t0:.0f}s avg_loss={avg:.4f}\")\n    return avg","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.255003Z","iopub.execute_input":"2025-11-10T15:22:53.255260Z","iopub.status.idle":"2025-11-10T15:22:53.267280Z","shell.execute_reply.started":"2025-11-10T15:22:53.255239Z","shell.execute_reply":"2025-11-10T15:22:53.266706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def validate_and_map(model, loader):\n    model.eval()\n    criterion = nn.BCEWithLogitsLoss()\n    running_loss = 0.0\n    n = 0\n    actuals = []\n    predicteds = []\n    with torch.no_grad():\n        for x, y in loader:\n            x = x.to(DEVICE)\n            y = y.to(DEVICE)\n            logits = model(x)\n            loss = criterion(logits, y)\n            running_loss += loss.item() * x.size(0)\n            n += x.size(0)\n            probs = torch.sigmoid(logits).cpu().numpy()\n            for row_idx in range(probs.shape[0]):\n                p = probs[row_idx]\n                topk = np.argsort(p)[-3:][::-1]\n                predicted_labels = [LABELS[i] for i in topk]\n                predicteds.append(predicted_labels)\n            y_np = y.cpu().numpy()\n            for row_idx in range(y_np.shape[0]):\n                gt_idx = np.where(y_np[row_idx] > 0.5)[0].tolist()\n                if len(gt_idx) == 0:\n                    actuals.append([])\n                else:\n                    actuals.append([LABELS[i] for i in gt_idx])\n    avg_loss = running_loss / n\n    map3 = mapk(actuals, predicteds, k=3)\n    print(f\"Validation loss {avg_loss:.4f} MAP@3 {map3:.4f}\")\n    return avg_loss, map3\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.267941Z","iopub.execute_input":"2025-11-10T15:22:53.268167Z","iopub.status.idle":"2025-11-10T15:22:53.282440Z","shell.execute_reply.started":"2025-11-10T15:22:53.268141Z","shell.execute_reply":"2025-11-10T15:22:53.281698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_test_and_write(model, loader, out_path=\"submission.csv\"):\n    model.eval()\n    rows = []\n    with torch.no_grad():\n        for x, fnames in loader:\n            x = x.to(DEVICE)\n            logits = model(x)\n            probs = torch.sigmoid(logits).cpu().numpy()\n            for i in range(probs.shape[0]):\n                p = probs[i]\n                topk = np.argsort(p)[-3:][::-1]\n                labels_pred = [LABELS[idx] for idx in topk]\n                rows.append((fnames[i], \" \".join(labels_pred)))\n    sub = pd.DataFrame(rows, columns=[\"fname\", \"label\"])\n    sub.to_csv(out_path, index=False)\n    print(\"Wrote\", out_path, \"rows:\", len(sub))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.283072Z","iopub.execute_input":"2025-11-10T15:22:53.283300Z","iopub.status.idle":"2025-11-10T15:22:53.298004Z","shell.execute_reply.started":"2025-11-10T15:22:53.283260Z","shell.execute_reply":"2025-11-10T15:22:53.297263Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def main():\n    tr_df, val_df = train_test_split(train_df, test_size=0.1, random_state=SEED, shuffle=True, stratify=None)\n    tr_df = tr_df.reset_index(drop=True)\n    val_df = val_df.reset_index(drop=True)\n    print(\"Train samples:\", len(tr_df), \"Val samples:\", len(val_df), \"Test samples:\", len(test_df))\n\n    train_ds = FreesoundDataset(tr_df, train_audio_dir, is_test=False)\n    val_ds   = FreesoundDataset(val_df, train_audio_dir, is_test=False)\n    test_ds  = FreesoundDataset(test_df, test_audio_dir, is_test=True)\n\n    train_loader = DataLoader(train_ds, batch_size=BATCH_SIZE, shuffle=True, num_workers=NUM_WORKERS, pin_memory=True)\n    val_loader   = DataLoader(val_ds, batch_size=BATCH_SIZE, shuffle=False, num_workers=NUM_WORKERS, pin_memory=True)\n    test_loader  = DataLoader(test_ds, batch_size=BATCH_SIZE, shuffle=False, num_workers=NUM_WORKERS, pin_memory=True)\n\n    model = AudioCNN(n_classes=NUM_CLASSES, in_ch=1).to(DEVICE)\n    optimizer = torch.optim.Adam(model.parameters(), lr=LR, weight_decay=1e-5)\n    scheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=2, verbose=True)\n    scaler = GradScaler()\n\n    best_map3 = -1.0\n    best_epoch = -1\n    for epoch in range(1, EPOCHS + 1):\n        train_loss = train_one_epoch(model, train_loader, optimizer, scaler, epoch)\n        val_loss, val_map3 = validate_and_map(model, val_loader)\n        scheduler.step(val_loss)\n        # save best by map3\n        if val_map3 > best_map3:\n            best_map3 = val_map3\n            best_epoch = epoch\n            torch.save(model.state_dict(), \"best_model.pth\")\n            print(f\"Saved best_model.pth (epoch {epoch}) MAP@3={val_map3:.4f}\")\n    print(\"Training finished. Best epoch:\", best_epoch, \"best MAP@3:\", best_map3)\n\n    # load best and predict test\n    if os.path.exists(\"best_model.pth\"):\n        model.load_state_dict(torch.load(\"best_model.pth\", map_location=DEVICE))\n    predict_test_and_write(model, test_loader, out_path=\"submission.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.298707Z","iopub.execute_input":"2025-11-10T15:22:53.299202Z","iopub.status.idle":"2025-11-10T15:22:53.310709Z","shell.execute_reply.started":"2025-11-10T15:22:53.299168Z","shell.execute_reply":"2025-11-10T15:22:53.310155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"if __name__ == \"__main__\":\n    main()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-10T15:22:53.311408Z","iopub.execute_input":"2025-11-10T15:22:53.311626Z","iopub.status.idle":"2025-11-10T15:42:54.924302Z","shell.execute_reply.started":"2025-11-10T15:22:53.311606Z","shell.execute_reply":"2025-11-10T15:42:54.923296Z"}},"outputs":[],"execution_count":null}]}