{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":8900,"databundleVersionId":862232,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"3ac3eeea-b0c1-4cad-b3ea-99459e86b6a2","cell_type":"code","source":"import os, random, math\nimport numpy as np\nimport pandas as pd\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nSEED=42\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntorch.manual_seed(SEED)\ntorch.cuda.manual_seed_all(SEED)\n\ndevice=torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nDATA_ROOT=\"/kaggle/input\"\nDATA_DIR=None\nfor d in os.listdir(DATA_ROOT):\n    p=os.path.join(DATA_ROOT,d)\n    if os.path.isdir(p) and os.path.exists(os.path.join(p,\"train.csv\")) and os.path.exists(os.path.join(p,\"audio_train\")):\n        DATA_DIR=p\n        break\nif DATA_DIR is None:\n    DATA_DIR=DATA_ROOT\n\nTRAIN_AUDIO_DIR=os.path.join(DATA_DIR,\"audio_train\")\nTEST_AUDIO_DIR=os.path.join(DATA_DIR,\"audio_test\")\nTRAIN_CSV=os.path.join(DATA_DIR,\"train.csv\")\nTRAIN_POST_CSV=os.path.join(DATA_DIR,\"train_post_competition.csv\")\nSAMPLE_SUB=os.path.join(DATA_DIR,\"sample_submission.csv\")\n\ntrain_df=pd.read_csv(TRAIN_CSV)\nif os.path.exists(TRAIN_POST_CSV):\n    t2=pd.read_csv(TRAIN_POST_CSV)\n    if \"fname\" in t2.columns and \"label\" in t2.columns:\n        train_df=pd.concat([train_df,t2],ignore_index=True)\n\nsub_df=pd.read_csv(SAMPLE_SUB)\n\nlabels=sorted(train_df[\"label\"].unique())\nlabel2idx={l:i for i,l in enumerate(labels)}\nidx2label={i:l for l,i in label2idx.items()}\ntrain_df[\"label_idx\"]=train_df[\"label\"].map(label2idx).astype(int)\n\nfrom sklearn.model_selection import StratifiedShuffleSplit\nsss=StratifiedShuffleSplit(n_splits=1,test_size=0.15,random_state=SEED)\ntr_idx,va_idx=next(sss.split(train_df,train_df[\"label_idx\"]))\ntr_df=train_df.iloc[tr_idx].reset_index(drop=True)\nva_df=train_df.iloc[va_idx].reset_index(drop=True)\n\nlen(labels),len(tr_df),len(va_df),str(device)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T21:19:07.827265Z","iopub.execute_input":"2025-12-15T21:19:07.82754Z","iopub.status.idle":"2025-12-15T21:19:10.380373Z","shell.execute_reply.started":"2025-12-15T21:19:07.827522Z","shell.execute_reply":"2025-12-15T21:19:10.379651Z"}},"outputs":[],"execution_count":null},{"id":"4a931adf-2487-4e14-a4ef-d0e8341b22f0","cell_type":"code","source":"import librosa\ntry:\n    import soundfile as sf\nexcept Exception:\n    sf=None\ntry:\n    from scipy.signal import resample_poly\nexcept Exception:\n    resample_poly=None\n\nTARGET_SR=32000\nDURATION=3.0\nN_FFT=1024\nHOP=256\nN_MELS=128\nFMIN=20\nFMAX=TARGET_SR//2\nFRAMES=256\n\nBATCH=32\nEPOCHS=14\nLR=3e-4\n\ndef _resample_linear(y, orig_sr, target_sr):\n    if orig_sr == target_sr:\n        return y.astype(np.float32, copy=False)\n    x_old = np.linspace(0.0, 1.0, num=y.shape[0], endpoint=False)\n    n_new = int(round(y.shape[0] * float(target_sr) / float(orig_sr)))\n    x_new = np.linspace(0.0, 1.0, num=n_new, endpoint=False)\n    y_new = np.interp(x_new, x_old, y).astype(np.float32, copy=False)\n    return y_new\n\ndef load_audio(path, target_sr=TARGET_SR, seconds=DURATION):\n    if sf is not None:\n        y, sr = sf.read(path, always_2d=False)\n        if isinstance(y, np.ndarray) and y.ndim==2:\n            y = y.mean(axis=1)\n        y = y.astype(np.float32, copy=False)\n    else:\n        y, sr = librosa.load(path, sr=None, mono=True)\n        y = y.astype(np.float32, copy=False)\n    if (target_sr is not None) and (sr != target_sr):\n        if resample_poly is not None:\n            g = math.gcd(sr, target_sr)\n            up = target_sr // g\n            down = sr // g\n            y = resample_poly(y, up, down).astype(np.float32, copy=False)\n        else:\n            y = _resample_linear(y, sr, target_sr)\n        sr = target_sr\n    need = int(sr * seconds)\n    if y.shape[0] < need:\n        y = np.pad(y, (0, need - y.shape[0]), mode=\"constant\")\n    elif y.shape[0] > need:\n        s = (y.shape[0] - need)//2\n        y = y[s:s+need]\n    return y, sr\n\ndef logmel(y, sr=TARGET_SR):\n    m = librosa.feature.melspectrogram(\n        y=y, sr=sr, n_fft=N_FFT, hop_length=HOP,\n        n_mels=N_MELS, fmin=FMIN, fmax=FMAX, power=2.0\n    )\n    m = librosa.power_to_db(m, ref=np.max)\n    return m\n\ndef pad_or_crop_2d(x, frames=FRAMES, random_crop=False):\n    T = x.shape[1]\n    if T < frames:\n        pad = frames - T\n        v = float(x.min()) if x.size else -80.0\n        return np.pad(x, ((0,0),(0,pad)), mode=\"constant\", constant_values=v)\n    if random_crop:\n        s = np.random.randint(0, T - frames + 1)\n    else:\n        s = max(0, (T - frames)//2)\n    return x[:, s:s+frames]\n\ndef make_3ch(spec):\n    d1 = librosa.feature.delta(spec)\n    d2 = librosa.feature.delta(spec, order=2)\n    x = np.stack([spec, d1, d2], axis=0).astype(np.float32)\n    mu = x.mean()\n    sd = x.std() + 1e-6\n    return (x - mu) / sd\n\ndef spec_augment(x, p=0.6, freq_mask=16, time_mask=32):\n    if random.random() > p:\n        return x\n    c,h,w = x.shape\n    f = random.randint(0, freq_mask)\n    f0 = random.randint(0, max(0, h-f)) if h-f>0 else 0\n    if f>0:\n        x[:, f0:f0+f, :] = 0\n    t = random.randint(0, time_mask)\n    t0 = random.randint(0, max(0, w-t)) if w-t>0 else 0\n    if t>0:\n        x[:, :, t0:t0+t] = 0\n    return x\n\ndef map3(y_true, probs, k=3):\n    topk = np.argsort(-probs, axis=1)[:, :k]\n    s = 0.0\n    for i, t in enumerate(y_true):\n        hit = np.where(topk[i] == t)[0]\n        if len(hit):\n            s += 1.0/(hit[0]+1.0)\n    return s/len(y_true)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T21:19:10.381735Z","iopub.execute_input":"2025-12-15T21:19:10.38218Z","iopub.status.idle":"2025-12-15T21:19:10.414293Z","shell.execute_reply.started":"2025-12-15T21:19:10.382158Z","shell.execute_reply":"2025-12-15T21:19:10.413709Z"}},"outputs":[],"execution_count":null},{"id":"904a0a17-69da-4856-a2b3-bdb74cfdd3ba","cell_type":"code","source":"model = nn.Sequential(\n    nn.Conv2d(3,32,3,padding=1,bias=False),\n    nn.BatchNorm2d(32),\n    nn.SiLU(),\n    nn.MaxPool2d(2),\n    nn.Dropout(0.15),\n\n    nn.Conv2d(32,64,3,padding=1,bias=False),\n    nn.BatchNorm2d(64),\n    nn.SiLU(),\n    nn.MaxPool2d(2),\n    nn.Dropout(0.20),\n\n    nn.Conv2d(64,128,3,padding=1,bias=False),\n    nn.BatchNorm2d(128),\n    nn.SiLU(),\n    nn.MaxPool2d(2),\n    nn.Dropout(0.25),\n\n    nn.Conv2d(128,192,3,padding=1,bias=False),\n    nn.BatchNorm2d(192),\n    nn.SiLU(),\n    nn.AdaptiveAvgPool2d((1,1)),\n\n    nn.Flatten(),\n    nn.Dropout(0.35),\n    nn.Linear(192, len(labels))\n).to(device)\n\nopt = torch.optim.AdamW(model.parameters(), lr=LR, weight_decay=1e-4)\ncrit = nn.CrossEntropyLoss(reduction=\"none\", label_smoothing=0.05)\nscaler = torch.cuda.amp.GradScaler(enabled=(device.type==\"cuda\"))\n\nsum(p.numel() for p in model.parameters())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T21:19:10.414894Z","iopub.execute_input":"2025-12-15T21:19:10.415211Z","iopub.status.idle":"2025-12-15T21:19:11.66432Z","shell.execute_reply.started":"2025-12-15T21:19:10.415193Z","shell.execute_reply":"2025-12-15T21:19:11.663482Z"}},"outputs":[],"execution_count":null},{"id":"1a6c71ee-c653-4be5-bf5a-002ac3d86176","cell_type":"code","source":"def get_batch(df, order, start, bs, training):\n    xs=[]\n    ys=[]\n    ws=[]\n    end=min(start+bs, len(order))\n    for j in range(start, end):\n        i = int(order[j])\n        r = df.iloc[i]\n        path = os.path.join(TRAIN_AUDIO_DIR, r[\"fname\"])\n        y, sr = load_audio(path)\n        sp = logmel(y, sr)\n        sp = pad_or_crop_2d(sp, FRAMES, random_crop=training)\n        x = make_3ch(sp)\n        if training:\n            x = spec_augment(x)\n        mv = int(r[\"manually_verified\"]) if (\"manually_verified\" in r and not pd.isna(r[\"manually_verified\"])) else 0\n        w = 1.0 if mv==1 else 0.35\n        xs.append(torch.tensor(x, dtype=torch.float32))\n        ys.append(int(r[\"label_idx\"]))\n        ws.append(w)\n    X = torch.stack(xs, dim=0)\n    Y = torch.tensor(ys, dtype=torch.long)\n    W = torch.tensor(ws, dtype=torch.float32)\n    return X, Y, W\n\nbest=0.0\nbest_path=\"best.pt\"\n\nfor epoch in range(EPOCHS):\n    model.train()\n    order = np.random.permutation(len(tr_df))\n    for start in range(0, len(order), BATCH):\n        X,Y,W = get_batch(tr_df, order, start, BATCH, training=True)\n        X = X.to(device, non_blocking=True)\n        Y = Y.to(device, non_blocking=True)\n        W = W.to(device, non_blocking=True)\n\n        opt.zero_grad(set_to_none=True)\n        with torch.cuda.amp.autocast(enabled=(device.type==\"cuda\")):\n            logits = model(X)\n            loss = (crit(logits, Y) * W).mean()\n        scaler.scale(loss).backward()\n        scaler.step(opt)\n        scaler.update()\n\n    model.eval()\n    probs_all=[]\n    y_all=[]\n    with torch.no_grad():\n        order2 = np.arange(len(va_df))\n        for start in range(0, len(order2), BATCH):\n            X,Y,W = get_batch(va_df, order2, start, BATCH, training=False)\n            X = X.to(device, non_blocking=True)\n            logits = model(X)\n            probs = torch.softmax(logits, dim=1).detach().cpu().numpy()\n            probs_all.append(probs)\n            y_all.append(Y.numpy())\n    probs_all = np.concatenate(probs_all, axis=0)\n    y_all = np.concatenate(y_all, axis=0)\n    score = float(map3(y_all, probs_all, 3))\n    if score > best:\n        best = score\n        torch.save(model.state_dict(), best_path)\n    print(epoch+1, score, best)\n\nbest\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T21:19:11.665875Z","iopub.execute_input":"2025-12-15T21:19:11.666239Z","iopub.status.idle":"2025-12-15T23:11:53.763763Z","shell.execute_reply.started":"2025-12-15T21:19:11.666221Z","shell.execute_reply":"2025-12-15T23:11:53.762788Z"}},"outputs":[],"execution_count":null},{"id":"d9d7ae1a-f421-4421-80f3-5ab3e530d6d7","cell_type":"code","source":"model.load_state_dict(torch.load(\"best.pt\", map_location=device))\nmodel.eval()\n\ndef predict_one(fname):\n    path = os.path.join(TEST_AUDIO_DIR, fname)\n    y, sr = load_audio(path)\n    sp = logmel(y, sr)\n    sp = pad_or_crop_2d(sp, FRAMES, random_crop=False)\n    x = make_3ch(sp)\n    X = torch.tensor(x, dtype=torch.float32)[None, ...].to(device)\n    with torch.no_grad():\n        p = torch.softmax(model(X), dim=1).squeeze(0).detach().cpu().numpy()\n    return p\n\npreds=[]\nfor fname in sub_df[\"fname\"].values:\n    p = predict_one(fname)\n    top3 = np.argsort(-p)[:3]\n    preds.append(\" \".join([idx2label[int(i)] for i in top3]))\n\nsubmission = pd.DataFrame({\"fname\": sub_df[\"fname\"].values, \"label\": preds})\nsubmission.to_csv(\"submission.csv\", index=False)\nsubmission.head()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T23:11:53.764835Z","iopub.execute_input":"2025-12-15T23:11:53.76508Z","iopub.status.idle":"2025-12-15T23:16:00.872141Z","shell.execute_reply.started":"2025-12-15T23:11:53.765058Z","shell.execute_reply":"2025-12-15T23:16:00.871418Z"}},"outputs":[],"execution_count":null}]}