{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":113558,"databundleVersionId":14878066,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"In the previous notebook, we did phase 1 of the solution where we used image classifiation algorithm to classify the image into **authentic** or **forged**. \n\nLink to the previous Notebook : https://www.kaggle.com/code/tany1404/recod-ai-luc-base-model-phase-1-v0\n\nIn this note we will implement the next phase where we want to find the mask for the solution. Here we use some pretrained model for the segmentation task. Right now for the simplicity, for training of this model we will use train_images/forged and train_masks.","metadata":{}},{"cell_type":"markdown","source":"## Imports and Initial Configuration","metadata":{}},{"cell_type":"code","source":"import pathlib\nimport numpy as np\nimport pandas as pd\nimport torch \nfrom PIL import Image\nimport cv2\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nimport torch.nn as nn \nfrom torch.utils.data import Dataset, DataLoader\nimport torch.functional as F\nimport torchvision.models as models\nfrom torchvision.transforms import transforms\n\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-24T05:17:58.607671Z","iopub.execute_input":"2025-12-24T05:17:58.607946Z","iopub.status.idle":"2025-12-24T05:17:58.613021Z","shell.execute_reply.started":"2025-12-24T05:17:58.607930Z","shell.execute_reply":"2025-12-24T05:17:58.612373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DEVICE = 'cuda' if torch.cuda.is_available() else 'cpu'\nprint(f\"using {DEVICE} device\")\nEPOCHS =  10\nBASE_DIR = pathlib.Path(\"/kaggle/input/recodai-luc-scientific-image-forgery-detection\")\nTRAIN_IMG_DIR = BASE_DIR / \"train_images\"\nTRAIN_IMG_MASK = BASE_DIR / \"train_masks\"\nFORGED_DIR = TRAIN_IMG_DIR / 'forged'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T05:17:58.614138Z","iopub.execute_input":"2025-12-24T05:17:58.614382Z","iopub.status.idle":"2025-12-24T05:17:58.627879Z","shell.execute_reply.started":"2025-12-24T05:17:58.614358Z","shell.execute_reply":"2025-12-24T05:17:58.627276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_PATH_ARR = []\n\nfor i in FORGED_DIR.iterdir():\n    tmp = {'img_path': i}\n    img_id = i.stem\n    mask_path = TRAIN_IMG_MASK / f\"{img_id}.npy\"\n    \n    if not mask_path.exists():\n        print(f\"mask for id {img_id} is not present\")\n        continue\n        \n    tmp['mask_path'] = mask_path\n    IMG_PATH_ARR.append(tmp)\n    \nDATA = pd.DataFrame(IMG_PATH_ARR).sample(frac=1).reset_index(drop=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T05:17:58.628609Z","iopub.execute_input":"2025-12-24T05:17:58.628884Z","iopub.status.idle":"2025-12-24T05:18:00.752022Z","shell.execute_reply.started":"2025-12-24T05:17:58.628869Z","shell.execute_reply":"2025-12-24T05:18:00.751445Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## creating Dataset annd DataLoader","metadata":{}},{"cell_type":"code","source":"train_df, test_df = train_test_split(DATA, test_size=0.2, random_state=42)\n\ntrain_df.shape, test_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T05:18:00.752745Z","iopub.execute_input":"2025-12-24T05:18:00.752954Z","iopub.status.idle":"2025-12-24T05:18:00.759905Z","shell.execute_reply.started":"2025-12-24T05:18:00.752937Z","shell.execute_reply":"2025-12-24T05:18:00.759216Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class LUCSegmentationDataclass(Dataset):\n    def __init__(self, df, img_transform=None, mask_transform=None):\n        self.df = df\n        self.img_transform = img_transform\n        self.mask_transform = mask_transform\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n        img = cv2.imread(row.img_path, cv2.IMREAD_GRAYSCALE)\n        mask = np.load(row.mask_path)\n        # mask = np.array(mask, dtype=np.float32)[0]\n        mask = np.argmax(mask, axis=0)\n\n        #print(img.shape, mask.shape)\n        if(self.img_transform):\n            img = self.transform(img)\n        if(self.mask_transform):\n            mask = self.transform(mask)\n        return img, mask\n\n\n    def __len__(self):\n        return len(self.df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T05:18:00.761942Z","iopub.execute_input":"2025-12-24T05:18:00.762195Z","iopub.status.idle":"2025-12-24T05:18:00.775341Z","shell.execute_reply.started":"2025-12-24T05:18:00.762180Z","shell.execute_reply":"2025-12-24T05:18:00.774746Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Model building","metadata":{}},{"cell_type":"code","source":"class LUCSegmentationDataclass(Dataset):\n    def __init__(self, df, img_transform=None):\n        self.df = df\n        self.img_transform = img_transform\n\n    def __getitem__(self, idx):\n        row = self.df.iloc[idx]\n\n        img = cv2.imread(row.img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        mask = np.load(row.mask_path)\n        mask = np.argmax(mask, axis=0)  # (H, W)\n        mask = (mask > 0).astype(np.int64)\n\n        if self.img_transform:\n            augmented = self.img_transform(image=img, mask=mask)\n            img = augmented[\"image\"]\n            mask = augmented[\"mask\"]\n        # else:\n            # img = torch.from_numpy(img).permute(2, 0, 1).float() / 255.0\n            # mask = torch.from_numpy(mask)\n        return img, mask.long()\n\n    def __len__(self):\n        return len(self.df)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T05:18:00.775978Z","iopub.execute_input":"2025-12-24T05:18:00.776266Z","iopub.status.idle":"2025-12-24T05:18:00.793881Z","shell.execute_reply.started":"2025-12-24T05:18:00.776245Z","shell.execute_reply":"2025-12-24T05:18:00.793299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transform = A.Compose([\n    A.Resize(256, 256),\n    A.HorizontalFlip(p=0.5),\n    A.RandomBrightnessContrast(p=0.3),\n    \n    A.Normalize(\n    mean=[0.485, 0.456, 0.406],\n    std=[0.229, 0.224, 0.225]\n    ),   \n    ToTensorV2(),\n])\n\ntest_transform = A.Compose([\n    A.Resize(256, 256),    \n    A.Normalize(\n    mean=[0.485, 0.456, 0.406],\n    std=[0.229, 0.224, 0.225]\n    ), \n    ToTensorV2(),\n])\n\n\ntrain_dataset = LUCSegmentationDataclass(train_df, transform)\ntest_dataset = LUCSegmentationDataclass(test_df,test_transform)\n\ntrain_dataloader = DataLoader(train_dataset, batch_size=16, shuffle=True)\ntest_dataloader = DataLoader(test_dataset, batch_size=16)\n\n# for img, mask in train_dataloader:\n#     print(img.shape, mask.shape)\n#     break\n# for img, mask in test_dataloader:\n#     print(img.shape, mask.shape)\n#     break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T05:18:00.794493Z","iopub.execute_input":"2025-12-24T05:18:00.795172Z","iopub.status.idle":"2025-12-24T05:18:00.813836Z","shell.execute_reply.started":"2025-12-24T05:18:00.795156Z","shell.execute_reply":"2025-12-24T05:18:00.813145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# DeepLabV3 expects:\n# Image: tensor of shape (C,H,W), float32, normalized.\n# Mask: tensor of shape (H,W), long dtype.\n\nclass LUCSegmentationModel(nn.Module):\n    def __init__(self, num_class):\n        super().__init__()\n        self.model = models.segmentation.deeplabv3_resnet50(pretrained=True)\n        self.model.classifier[4] = nn.Conv2d(256, num_class, kernel_size=1)\n\n        # Copy pretrained weights: take mean over RGB channels\n        with torch.no_grad():\n            for p in self.model.backbone.parameters():\n                p.requires_grad = False\n    \n            for p in self.model.backbone.layer4.parameters():\n                p.requires_grad = True\n    \n    def forward(self, x):\n        out = self.model(x)[\"out\"]\n        return out","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T05:18:00.814658Z","iopub.execute_input":"2025-12-24T05:18:00.814981Z","iopub.status.idle":"2025-12-24T05:18:00.823665Z","shell.execute_reply.started":"2025-12-24T05:18:00.814958Z","shell.execute_reply":"2025-12-24T05:18:00.823073Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_SEGMENTATION_MODEL = LUCSegmentationModel(2).to(DEVICE)\n\nLOSS_FN = nn.CrossEntropyLoss()\n\nOPTIMIZER = torch.optim.Adam(BASE_SEGMENTATION_MODEL.parameters(), lr=0.01)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T05:18:00.824227Z","iopub.execute_input":"2025-12-24T05:18:00.824451Z","iopub.status.idle":"2025-12-24T05:18:01.485639Z","shell.execute_reply.started":"2025-12-24T05:18:00.824431Z","shell.execute_reply":"2025-12-24T05:18:01.485078Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training of the model","metadata":{}},{"cell_type":"code","source":"TOTAL_TRAIN_LOSS = []\nTOTAL_TEST_LOSS = []\n\nfor epoch in range(EPOCHS):\n\n    # training loop\n    tmp_train_loss = []\n    for img, mask in train_dataloader:\n        img = img.to(DEVICE)\n        mask = mask.to(DEVICE)\n\n        OPTIMIZER.zero_grad()\n        pred = BASE_SEGMENTATION_MODEL(img)\n        loss = LOSS_FN(pred, mask)\n\n        loss.backward()\n        OPTIMIZER.step()\n        tmp_train_loss.append(loss.item())\n    TOTAL_TRAIN_LOSS.append(np.mean(tmp_train_loss))\n    \n    # testing loop\n    BASE_SEGMENTATION_MODEL.eval()\n    tmp_test_loss = []\n    with torch.no_grad():\n        for img, mask in test_dataloader:\n            img = img.to(DEVICE)\n            mask = mask.to(DEVICE)\n            pred = BASE_SEGMENTATION_MODEL(img)\n            loss = LOSS_FN(pred, mask)\n            tmp_test_loss.append(loss.item())\n    TOTAL_TEST_LOSS.append(np.mean(tmp_test_loss))\n\n    print(f\"Epoch {epoch+1}/{EPOCHS} → Train Loss: {TOTAL_TRAIN_LOSS[-1]:.4f}, Test Loss: {TOTAL_TEST_LOSS[-1]:.4f}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T05:18:01.486502Z","iopub.execute_input":"2025-12-24T05:18:01.486787Z","iopub.status.idle":"2025-12-24T05:55:13.022320Z","shell.execute_reply.started":"2025-12-24T05:18:01.486758Z","shell.execute_reply":"2025-12-24T05:55:13.021604Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 6))\n\n\nplt.plot(np.arange(EPOCHS), TOTAL_TRAIN_LOSS, label=\"Train Loss\",linewidth=2)\n\nplt.plot(np.arange(EPOCHS), TOTAL_TEST_LOSS, label=\"Test Loss\",\n    linewidth=2\n)\n\n# Labels and title\nplt.xlabel(\"Epochs\", fontsize=12)\nplt.ylabel(\"Loss\", fontsize=12)\nplt.title(\"Train Loss vs Test Loss Over Epochs\", fontsize=14)\n\n# Show epoch markers\nplt.xticks(np.arange(EPOCHS))\n\n# Grid for clarity\nplt.grid(True, linestyle=\"--\", alpha=0.5)\n\n# Legend\nplt.legend(fontsize=12)\n\n# Tight layout for better spacing\nplt.tight_layout()\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T05:55:13.023073Z","iopub.execute_input":"2025-12-24T05:55:13.023329Z","iopub.status.idle":"2025-12-24T05:55:13.245471Z","shell.execute_reply.started":"2025-12-24T05:55:13.023306Z","shell.execute_reply":"2025-12-24T05:55:13.244792Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Testing","metadata":{}},{"cell_type":"code","source":"TEST_DIR = BASE_DIR / \"test_images\"\n\nBASE_SEGMENTATION_MODEL.eval()\nfor img_path in TEST_DIR.iterdir():\n    img = Image.open(img_path).convert(\"RGB\")\n    to_tensor = transforms.ToTensor()\n\n    img_tensor = to_tensor(img).to(DEVICE).unsqueeze(dim=0)\n    pred_val = BASE_SEGMENTATION_MODEL(img_tensor)\n    print(pred_val.shape, img_tensor.shape)\n\n    mask = torch.argmax(pred_val, dim=1).squeeze(0) \n\n    plt.figure(figsize=(10, 10)) \n    plt.imshow(mask.cpu(), cmap='jet') \n    plt.title(\"Segmentation Mask\") \n    plt.axis(\"off\") \n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T05:55:13.246722Z","iopub.execute_input":"2025-12-24T05:55:13.246913Z","iopub.status.idle":"2025-12-24T05:55:14.238777Z","shell.execute_reply.started":"2025-12-24T05:55:13.246898Z","shell.execute_reply":"2025-12-24T05:55:14.238000Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}