{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":113558,"databundleVersionId":14878066,"sourceType":"competition"},{"sourceId":2214915,"sourceType":"datasetVersion","datasetId":1330115},{"sourceId":6699845,"sourceType":"datasetVersion","datasetId":3861756},{"sourceId":47589828,"sourceType":"kernelVersion"}],"dockerImageVersionId":31236,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# ============================================================\n# RECOD.AI LUC — FINAL REALISTIC BASELINE (CLASSIFICATION)\n# ============================================================\n\nimport os, glob, random\nimport numpy as np\nimport pandas as pd\nfrom pathlib import Path\nfrom tqdm import tqdm\n\nimport cv2\nimport torch\nimport torch.nn as nn\nfrom torch.utils.data import Dataset, DataLoader\n\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport timm\n\n# ================= CONFIG =================\nclass CFG:\n    ROOT = \"/kaggle/input/recod-ai-luc-scientific-image-forgery-detection\"\n    IMG_SIZE = 224\n    BS = 16\n    EPOCHS = 10\n    LR = 3e-4\n    DEVICE = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\n# ================= SEED =================\ndef seed_all(seed=42):\n    random.seed(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n\nseed_all()\n\n# ================= DATA =================\ndef build_df():\n    rows = []\n\n    forged = glob.glob(f\"{CFG.ROOT}/train_images/forged/*.png\")\n    authentic = glob.glob(f\"{CFG.ROOT}/train_images/authentic/*.png\")\n\n    for p in forged:\n        rows.append({\"image\": p, \"label\": 1})\n    for p in authentic:\n        rows.append({\"image\": p, \"label\": 0})\n\n    df = pd.DataFrame(rows)\n    print(df.label.value_counts())\n\n    assert len(df) > 50, \"Dataset not found or paths wrong\"\n    return df\n\ndf = build_df()\n\n# ================= DATASET =================\nclass ForgeClsDS(Dataset):\n    def __init__(self, df, tfm):\n        self.df = df.reset_index(drop=True)\n        self.tfm = tfm\n\n    def __len__(self):\n        return len(self.df)\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img = cv2.imread(row.image)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img = self.tfm(image=img)[\"image\"]\n        return img, torch.tensor(row.label, dtype=torch.float32)\n\n# ================= TRANSFORMS =================\ntrain_tfm = A.Compose([\n    A.Resize(CFG.IMG_SIZE, CFG.IMG_SIZE),\n    A.HorizontalFlip(),\n    A.Normalize(),\n    ToTensorV2()\n])\n\nval_tfm = A.Compose([\n    A.Resize(CFG.IMG_SIZE, CFG.IMG_SIZE),\n    A.Normalize(),\n    ToTensorV2()\n])\n\n# ================= MODEL =================\nmodel = timm.create_model(\n    \"convnext_tiny\",\n    pretrained=False,   # offline safe\n    num_classes=1\n).to(CFG.DEVICE)\n\n# ================= TRAIN =================\nloader = DataLoader(\n    ForgeClsDS(df, train_tfm),\n    batch_size=CFG.BS,\n    shuffle=True\n)\n\nopt = torch.optim.AdamW(model.parameters(), CFG.LR)\ncriterion = nn.BCEWithLogitsLoss()\n\nfor e in range(CFG.EPOCHS):\n    model.train()\n    total = 0\n    for img, label in loader:\n        img = img.to(CFG.DEVICE)\n        label = label.to(CFG.DEVICE)\n\n        opt.zero_grad()\n        logits = model(img).squeeze(1)\n        loss = criterion(logits, label)\n        loss.backward()\n        opt.step()\n\n        total += loss.item()\n\n    print(f\"Epoch {e}: loss={total/len(loader):.4f}\")\n\n# ================= SUBMISSION =================\ntest_imgs = glob.glob(f\"{CFG.ROOT}/test_images/*.png\")\npreds = []\n\nmodel.eval()\nwith torch.no_grad():\n    for p in tqdm(test_imgs):\n        img = cv2.imread(p)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        img = val_tfm(image=img)[\"image\"].unsqueeze(0).to(CFG.DEVICE)\n\n        prob = torch.sigmoid(model(img)).item()\n        ann = \"authentic\" if prob < 0.5 else \"\"\n\n        preds.append({\n            \"case_id\": Path(p).stem,\n            \"annotation\": ann\n        })\n\npd.DataFrame(preds).to_csv(\"submission.csv\", index=False)\nprint(\"submission.csv saved\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-03T14:11:59.452326Z","iopub.execute_input":"2026-01-03T14:11:59.452598Z","iopub.status.idle":"2026-01-03T14:11:59.469959Z","shell.execute_reply.started":"2026-01-03T14:11:59.452575Z","shell.execute_reply":"2026-01-03T14:11:59.469145Z"}},"outputs":[{"traceback":["\u001b[0;31m---------------------------------------------------------------------------\u001b[0m","\u001b[0;31mAttributeError\u001b[0m                            Traceback (most recent call last)","\u001b[0;32m/tmp/ipykernel_55/1830382721.py\u001b[0m in \u001b[0;36m<cell line: 0>\u001b[0;34m()\u001b[0m\n\u001b[1;32m     54\u001b[0m     \u001b[0;32mreturn\u001b[0m \u001b[0mdf\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     55\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 56\u001b[0;31m \u001b[0mdf\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mbuild_df\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m     57\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     58\u001b[0m \u001b[0;31m# ================= DATASET =================\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/tmp/ipykernel_55/1830382721.py\u001b[0m in \u001b[0;36mbuild_df\u001b[0;34m()\u001b[0m\n\u001b[1;32m     49\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     50\u001b[0m     \u001b[0mdf\u001b[0m \u001b[0;34m=\u001b[0m \u001b[0mpd\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mDataFrame\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mrows\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m---> 51\u001b[0;31m     \u001b[0mprint\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mdf\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mlabel\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0mvalue_counts\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m     52\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m     53\u001b[0m     \u001b[0;32massert\u001b[0m \u001b[0mlen\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mdf\u001b[0m\u001b[0;34m)\u001b[0m \u001b[0;34m>\u001b[0m \u001b[0;36m50\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0;34m\"Dataset not found or paths wrong\"\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;32m/usr/local/lib/python3.12/dist-packages/pandas/core/generic.py\u001b[0m in \u001b[0;36m__getattr__\u001b[0;34m(self, name)\u001b[0m\n\u001b[1;32m   6297\u001b[0m         ):\n\u001b[1;32m   6298\u001b[0m             \u001b[0;32mreturn\u001b[0m \u001b[0mself\u001b[0m\u001b[0;34m[\u001b[0m\u001b[0mname\u001b[0m\u001b[0;34m]\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0;32m-> 6299\u001b[0;31m         \u001b[0;32mreturn\u001b[0m \u001b[0mobject\u001b[0m\u001b[0;34m.\u001b[0m\u001b[0m__getattribute__\u001b[0m\u001b[0;34m(\u001b[0m\u001b[0mself\u001b[0m\u001b[0;34m,\u001b[0m \u001b[0mname\u001b[0m\u001b[0;34m)\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n\u001b[0m\u001b[1;32m   6300\u001b[0m \u001b[0;34m\u001b[0m\u001b[0m\n\u001b[1;32m   6301\u001b[0m     \u001b[0;34m@\u001b[0m\u001b[0mfinal\u001b[0m\u001b[0;34m\u001b[0m\u001b[0;34m\u001b[0m\u001b[0m\n","\u001b[0;31mAttributeError\u001b[0m: 'DataFrame' object has no attribute 'label'"],"ename":"AttributeError","evalue":"'DataFrame' object has no attribute 'label'","output_type":"error"}],"execution_count":3}]}