{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":113558,"databundleVersionId":14174843,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport json\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom PIL import Image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:00.943138Z","iopub.execute_input":"2025-11-08T13:04:00.943895Z","iopub.status.idle":"2025-11-08T13:04:00.948727Z","shell.execute_reply.started":"2025-11-08T13:04:00.943866Z","shell.execute_reply":"2025-11-08T13:04:00.947629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rle_encode(mask):\n    \"\"\"\n    Convert binary mask to RLE string (standard Kaggle format: \"3 5 2 1\").\n    \"\"\"\n    pixels = mask.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:03.089083Z","iopub.execute_input":"2025-11-08T13:04:03.089777Z","iopub.status.idle":"2025-11-08T13:04:03.095580Z","shell.execute_reply.started":"2025-11-08T13:04:03.089746Z","shell.execute_reply":"2025-11-08T13:04:03.094390Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_mask(mask, title):\n    \"\"\"Visualize mask with pixel values and grid\"\"\"\n    plt.figure(figsize=(6, 6))\n    plt.imshow(mask, cmap='gray', vmin=0, vmax=1)\n    plt.title(title)\n    plt.axis('off')\n    \n    # Add grid\n    for i in range(mask.shape[0] + 1):\n        plt.axhline(i - 0.5, color='red', alpha=0.3, linewidth=0.5)\n        plt.axvline(i - 0.5, color='red', alpha=0.3, linewidth=0.5)\n    \n    # Show pixel values\n    for i in range(mask.shape[0]):\n        for j in range(mask.shape[1]):\n            plt.text(j, i, str(mask[i, j]), ha='center', va='center', \n                    color='blue' if mask[i, j] == 0 else 'white', fontweight='bold')\n    \n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:03.592401Z","iopub.execute_input":"2025-11-08T13:04:03.593339Z","iopub.status.idle":"2025-11-08T13:04:03.600520Z","shell.execute_reply.started":"2025-11-08T13:04:03.593303Z","shell.execute_reply":"2025-11-08T13:04:03.599241Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# [Code for visualize_mask function here]\ndef visualize_mask(mask, title):\n    \"\"\"Visualize mask with pixel values and grid\"\"\"\n    plt.figure(figsize=(6, 6))\n    plt.imshow(mask, cmap='gray', vmin=0, vmax=1)\n    plt.title(title)\n    plt.axis('off')\n    \n    # Add grid\n    for i in range(mask.shape[0] + 1):\n        plt.axhline(i - 0.5, color='red', alpha=0.3, linewidth=0.5)\n        plt.axvline(i - 0.5, color='red', alpha=0.3, linewidth=0.5)\n    \n    # Show pixel values\n    for i in range(mask.shape[0]):\n        for j in range(mask.shape[1]):\n            plt.text(j, i, str(mask[i, j]), ha='center', va='center', \n                    color='blue' if mask[i, j] == 0 else 'white', fontweight='bold')\n    \n    plt.show()\n\n# Corrected test mask definition:\ntest_mask = np.array([\n    [1, 0, 0], # Added commas between rows\n    [1, 1, 0],\n    [0, 1, 1],\n    [0, 0, 1]\n], dtype=np.uint8)\n\n# Visualize it\nvisualize_mask(test_mask, \"Example Binary Mask Visualization\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:11.649659Z","iopub.execute_input":"2025-11-08T13:04:11.650013Z","iopub.status.idle":"2025-11-08T13:04:11.799599Z","shell.execute_reply.started":"2025-11-08T13:04:11.649985Z","shell.execute_reply":"2025-11-08T13:04:11.798084Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# 1. Prepare data (e.g., an increasing trend)\nx_values = np.linspace(0, 10, 50)  # 50 points between 0 and 10\ny_values = np.sin(x_values) + np.random.normal(0, 0.1, 50) # Sine wave with some noise\n\n# 2. Create the plot\nplt.figure(figsize=(8, 4)) # Optional: set the figure size\nplt.plot(x_values, y_values, label='Sine Wave with Noise', color='blue', linestyle='-')\n\n# 3. Add labels, title, and a legend\nplt.xlabel('X Axis Value')\nplt.ylabel('Y Axis Value')\nplt.title('Simple Line Plot Example')\nplt.legend()\nplt.grid(True) # Optional: add a grid\n\n# 4. Display the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:16.029817Z","iopub.execute_input":"2025-11-08T13:04:16.030192Z","iopub.status.idle":"2025-11-08T13:04:16.231038Z","shell.execute_reply.started":"2025-11-08T13:04:16.030167Z","shell.execute_reply":"2025-11-08T13:04:16.229781Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# 1. Prepare data (random data for demonstration)\nnp.random.seed(42)\nx_data = np.random.rand(100) * 10\ny_data = 2 * x_data + np.random.rand(100) * 5 # A positive correlation\n\n# 2. Create the plot\nplt.figure(figsize=(6, 6))\nplt.scatter(x_data, y_data, color='red', marker='o', alpha=0.6)\n\n# 3. Add labels and title\nplt.xlabel('Variable X')\nplt.ylabel('Variable Y')\nplt.title('Scatter Plot: X vs Y')\n\n# 4. Display the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:18.896309Z","iopub.execute_input":"2025-11-08T13:04:18.897293Z","iopub.status.idle":"2025-11-08T13:04:19.088833Z","shell.execute_reply.started":"2025-11-08T13:04:18.897252Z","shell.execute_reply":"2025-11-08T13:04:19.087323Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# 1. Prepare data (e.g., normally distributed data)\ndata = np.random.randn(1000) # 1000 random values (mean 0, std dev 1)\n\n# 2. Create the plot\nplt.figure(figsize=(7, 4))\nplt.hist(data, bins=30, color='green', edgecolor='black', alpha=0.7)\n\n# 3. Add labels and title\nplt.xlabel('Value')\nplt.ylabel('Frequency')\nplt.title('Histogram of Random Data')\n\n# 4. Display the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:23.637310Z","iopub.execute_input":"2025-11-08T13:04:23.637736Z","iopub.status.idle":"2025-11-08T13:04:23.868128Z","shell.execute_reply.started":"2025-11-08T13:04:23.637711Z","shell.execute_reply":"2025-11-08T13:04:23.866851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# 1. Prepare dummy data for training metrics\nepochs = range(1, 11) # 10 epochs\ntrain_loss = [0.8, 0.6, 0.4, 0.3, 0.25, 0.21, 0.18, 0.16, 0.15, 0.14]\nval_loss = [0.75, 0.65, 0.5, 0.4, 0.35, 0.3, 0.28, 0.27, 0.26, 0.25]\ntrain_acc = [0.7, 0.75, 0.8, 0.85, 0.88, 0.9, 0.91, 0.92, 0.93, 0.935]\nval_acc = [0.72, 0.74, 0.78, 0.82, 0.84, 0.86, 0.87, 0.88, 0.89, 0.895]\n\n# 2. Create the figure and a set of subplots\n# (1 row, 2 columns) - returns the figure object and an array of axis objects (axs)\nfig, axs = plt.subplots(nrows=1, ncols=2, figsize=(12, 5))\n\n# --- Plot 1: Loss ---\naxs[0].plot(epochs, train_loss, label='Training Loss', color='blue')\naxs[0].plot(epochs, val_loss, label='Validation Loss', color='orange')\naxs[0].set_title('Model Loss Over Epochs')\naxs[0].set_xlabel('Epoch')\naxs[0].set_ylabel('Loss Value')\naxs[0].legend()\naxs[0].grid(True)\n\n# --- Plot 2: Accuracy ---\naxs[1].plot(epochs, train_acc, label='Training Accuracy', color='green')\naxs[1].plot(epochs, val_acc, label='Validation Accuracy', color='red')\naxs[1].set_title('Model Accuracy Over Epochs')\naxs[1].set_xlabel('Epoch')\naxs[1].set_ylabel('Accuracy Value')\naxs[1].legend()\naxs[1].grid(True)\n\n# 3. Add a super title for the entire figure and display\nfig.suptitle('Training Metrics Summary')\nplt.tight_layout(rect=[0, 0.03, 1, 0.95]) # Adjust layout to prevent title overlap\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:26.395869Z","iopub.execute_input":"2025-11-08T13:04:26.396234Z","iopub.status.idle":"2025-11-08T13:04:26.920347Z","shell.execute_reply.started":"2025-11-08T13:04:26.396211Z","shell.execute_reply":"2025-11-08T13:04:26.918922Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"simple_mask = np.array([\n    [1, 0],\n    [1, 1]\n])\nprint(f\"Mask:\\n{simple_mask}\")\nprint(f\"Flattened: {simple_mask.flatten()}\")\nprint(f\"RLE: '{rle_encode(simple_mask)}'\")\nvisualize_mask(simple_mask, \"Simple 2x2 Mask\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:33.246017Z","iopub.execute_input":"2025-11-08T13:04:33.246688Z","iopub.status.idle":"2025-11-08T13:04:33.376984Z","shell.execute_reply.started":"2025-11-08T13:04:33.246658Z","shell.execute_reply":"2025-11-08T13:04:33.376004Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport cv2 # Using OpenCV to load images if needed, or you can use PIL\n\n# 1. Prepare dummy data (you would load this from your dataset)\n# Using a simple red square in a blue image as an example\nimage = np.zeros((100, 100, 3), dtype=np.uint8)\nimage[:, :] = [0, 100, 200] # Blueish background\nimage[25:75, 25:75] = [255, 0, 0] # Red square\n\n# A binary mask for the red square\nmask = np.zeros((100, 100), dtype=np.uint8)\nmask[25:75, 25:75] = 1\n\n# 2. Create the figure and subplots (1 row, 2 columns)\nfig, axs = plt.subplots(nrows=1, ncols=2, figsize=(10, 5))\n\n# --- Plot 1: The Original Image ---\naxs[0].imshow(image)\naxs[0].set_title('Original Image')\naxs[0].axis('off') # Hide axis ticks\n\n# --- Plot 2: The Binary Mask ---\n# Use cmap='gray' for binary masks\naxs[1].imshow(mask, cmap='gray', vmin=0, vmax=1)\naxs[1].set_title('Binary Mask')\naxs[1].axis('off')\n\n# 3. Display the plots\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:34.428608Z","iopub.execute_input":"2025-11-08T13:04:34.428940Z","iopub.status.idle":"2025-11-08T13:04:34.748329Z","shell.execute_reply.started":"2025-11-08T13:04:34.428915Z","shell.execute_reply":"2025-11-08T13:04:34.747295Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport numpy as np\n\n# 1. Prepare dummy image data\nimage = np.zeros((200, 300, 3), dtype=np.uint8)\nimage[:, :] = [200, 200, 200] # Gray background\n\n# 2. Define a dummy bounding box [x, y, width, height]\n# x, y are the coordinates of the lower-left corner\nbbox = [50, 70, 100, 80]\n\n# 3. Create the figure and axes\nfig, ax = plt.subplots(1, figsize=(8, 6))\n\n# Display the image\nax.imshow(image)\n\n# Create a Rectangle patch for the bounding box\nrect = patches.Rectangle((bbox[0], bbox[1]), bbox[2], bbox[3],\n                         linewidth=2, edgecolor='r', facecolor='none')\n\n# Add the patch to the axes\nax.add_patch(rect)\n\n# Set title and remove axes\nax.set_title('Image with Bounding Box')\nax.axis('off')\n\n# Display the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:37.884793Z","iopub.execute_input":"2025-11-08T13:04:37.885144Z","iopub.status.idle":"2025-11-08T13:04:38.028586Z","shell.execute_reply.started":"2025-11-08T13:04:37.885123Z","shell.execute_reply":"2025-11-08T13:04:38.027474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay\nimport seaborn as sns # Optional, makes the plot look nicer\n\n# 1. Prepare dummy data (replace with your actual validation masks and predictions)\n# Example of flattened data:\ny_true = np.array([0, 0, 1, 1, 0, 1, 0, 1, 1, 0, 0, 0])\ny_pred = np.array([0, 1, 1, 0, 0, 1, 1, 1, 1, 0, 0, 0])\n\n# 2. Calculate the confusion matrix\ncm = confusion_matrix(y_true, y_pred, labels=[0, 1])\n# cm = [[TN, FP], [FN, TP]]\n\n# 3. Plot the confusion matrix using ConfusionMatrixDisplay\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm, display_labels=[ \"Background (0)\", \"Forgery (1)\"])\n\nplt.figure(figsize=(6, 6))\ndisp.plot(cmap=plt.cm.Blues, values_format='d') # use 'd' for integer counts\n\nplt.title('Confusion Matrix for Validation Set')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:38.875248Z","iopub.execute_input":"2025-11-08T13:04:38.875578Z","iopub.status.idle":"2025-11-08T13:04:39.089345Z","shell.execute_reply.started":"2025-11-08T13:04:38.875556Z","shell.execute_reply":"2025-11-08T13:04:39.088150Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nfrom sklearn.metrics import roc_curve, auc\n\n# 1. Prepare dummy data (replace with actual ground truth and predicted probabilities)\n# y_true are the binary labels (0 or 1)\n# y_probs are the raw probabilities (0.0 to 1.0)\ny_true = np.array([0, 0, 1, 1, 0, 1, 0, 1, 1, 0, 0, 0])\ny_probs = np.array([0.1, 0.6, 0.9, 0.2, 0.3, 0.8, 0.7, 0.95, 0.85, 0.15, 0.25, 0.05])\n\n# 2. Calculate the False Positive Rate (fpr), True Positive Rate (tpr), and thresholds\nfpr, tpr, thresholds = roc_curve(y_true, y_probs)\nroc_auc = auc(fpr, tpr) # Calculate the AUC score\n\n# 3. Plot the ROC curve\nplt.figure(figsize=(7, 6))\nplt.plot(fpr, tpr, color='darkorange', lw=2, label=f'ROC curve (AUC = {roc_auc:.2f})')\nplt.plot([0, 1], [0, 1], color='navy', lw=2, linestyle='--', label='Random Guessing (AUC = 0.50)')\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate (FPR)')\nplt.ylabel('True Positive Rate (TPR) / Recall')\nplt.title('Receiver Operating Characteristic (ROC) Curve')\nplt.legend(loc=\"lower right\")\nplt.grid(True)\n\n# 4. Display the plot\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:44.763146Z","iopub.execute_input":"2025-11-08T13:04:44.763471Z","iopub.status.idle":"2025-11-08T13:04:44.989744Z","shell.execute_reply.started":"2025-11-08T13:04:44.763448Z","shell.execute_reply":"2025-11-08T13:04:44.988602Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# Dummy data from previous example\nepochs = range(1, 11) \ntrain_loss = [0.8, 0.6, 0.4, 0.3, 0.25, 0.21, 0.18, 0.16, 0.15, 0.14]\n\n# Plot the loss using the horizontal line marker ('_')\nplt.figure(figsize=(8, 4))\nplt.plot(epochs, train_loss, label='Training Loss', color='blue', marker='_') # Use marker='_'\n\nplt.xlabel('Epoch')\nplt.ylabel('Loss Value')\nplt.title('Training Loss with Minus Markers')\nplt.legend()\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:51.872542Z","iopub.execute_input":"2025-11-08T13:04:51.872861Z","iopub.status.idle":"2025-11-08T13:04:52.080080Z","shell.execute_reply.started":"2025-11-08T13:04:51.872839Z","shell.execute_reply":"2025-11-08T13:04:52.078970Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\n# Create a large square mask (1s)\nlarge_square = np.ones((50, 50), dtype=np.uint8)\n\n# Create a small square mask inside it (0s inside 1s)\nsmall_square = np.zeros((50, 50), dtype=np.uint8)\nsmall_square[20:30, 20:30] = 1 # A 'hole' mask\n\n# Subtract the 'hole' mask to create the final shape\n# The final mask will have 1s where the large square was, but 0s in the center hole area.\nminus_shape_mask = large_square - small_square\n\n# Visualize the resulting mask\nplt.figure(figsize=(5, 5))\nplt.imshow(minus_shape_mask, cmap='gray', vmin=0, vmax=1)\nplt.title('Mask Created by Subtraction (Donut Shape)')\nplt.axis('off')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:04:54.913602Z","iopub.execute_input":"2025-11-08T13:04:54.913926Z","iopub.status.idle":"2025-11-08T13:04:55.013777Z","shell.execute_reply.started":"2025-11-08T13:04:54.913905Z","shell.execute_reply":"2025-11-08T13:04:55.012534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class EarlyStopping:\n    def __init__(self, patience=7, verbose=False, delta=0, path='checkpoint.pt'):\n        self.patience = patience\n        self.verbose = verbose\n        self.counter = 0\n        self.best_score = None\n        self.early_stop = False\n        self.val_loss_min = np.Inf\n        self.delta = delta\n        self.path = path\n\n    def __call__(self, val_loss, model):\n        score = -val_loss\n\n        if self.best_score is None:\n            self.best_score = score\n            self.save_checkpoint(val_loss, model)\n        elif score < self.best_score + self.delta:\n            self.counter += 1\n            if self.verbose:\n                print(f'EarlyStopping counter: {self.counter} out of {self.patience}')\n            if self.counter >= self.patience:\n                self.early_stop = True\n        else:\n            self.best_score = score\n            self.save_checkpoint(val_loss, model)\n            self.counter = 0\n\n    def save_checkpoint(self, val_loss, model):\n        if self.verbose:\n            print(f'Validation loss decreased ({self.val_loss_min:.6f} --> {val_loss:.6f}). Saving model...')\n        torch.save(model.state_dict(), self.path)\n        self.val_loss_min = val_loss\n\n# Usage in your training loop:\n# early_stopping = EarlyStopping(patience=10, verbose=True)\n# \n# for epoch in range(NUM_EPOCHS):\n#     # ... (run one training epoch) ...\n#     # ... (calculate validation loss for this epoch) ...\n#     \n#     early_stopping(val_loss_epoch, model)\n#     if early_stopping.early_stop:\n#         print(\"Early stopping triggered\")\n#         break\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:10:11.018931Z","iopub.execute_input":"2025-11-08T13:10:11.020192Z","iopub.status.idle":"2025-11-08T13:10:11.029580Z","shell.execute_reply.started":"2025-11-08T13:10:11.020153Z","shell.execute_reply":"2025-11-08T13:10:11.028444Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plus_mask = np.zeros((9, 9), dtype=np.uint8)\nplus_mask[2:7, 4] = 1  # Vertical line\nplus_mask[4, 2:7] = 1  # Horizontal line\n\nprint(\"Mask visualization:\")\nfor i in range(9):\n    print(' '.join(map(str, plus_mask[i])))\n\nprint(f\"Flattened (first 20): {' '.join(map(str, plus_mask.flatten()[:20]))}...\")\nprint(f\"RLE: '{rle_encode(plus_mask)}'\")\nvisualize_mask(plus_mask, \"Plus Shape Mask\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:10:15.283645Z","iopub.execute_input":"2025-11-08T13:10:15.283995Z","iopub.status.idle":"2025-11-08T13:10:15.519675Z","shell.execute_reply.started":"2025-11-08T13:10:15.283939Z","shell.execute_reply":"2025-11-08T13:10:15.518661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"minus_mask = np.zeros((9, 9), dtype=np.uint8)\nminus_mask[4, 2:7] = 1  # Horizontal line\n\nprint(\"Mask visualization:\")\nfor i in range(9):\n    print(' '.join(map(str, minus_mask[i])))\n\nprint(f\"RLE: '{rle_encode(minus_mask)}'\")\nvisualize_mask(minus_mask, \"Minus Shape Mask\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:10:16.502590Z","iopub.execute_input":"2025-11-08T13:10:16.503570Z","iopub.status.idle":"2025-11-08T13:10:16.737518Z","shell.execute_reply.started":"2025-11-08T13:10:16.503536Z","shell.execute_reply":"2025-11-08T13:10:16.736505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_edge_map(image_rgb):\n    # Convert RGB image to grayscale\n    gray = cv2.cvtColor(image_rgb, cv2.COLOR_RGB2GRAY)\n    # Use Canny edge detection\n    edges = cv2.Canny(gray, 100, 200)\n    # Normalize edges to be 0 or 1 float\n    edges = edges.astype(np.float32) / 255.0\n    return edges\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:10:18.966534Z","iopub.execute_input":"2025-11-08T13:10:18.966846Z","iopub.status.idle":"2025-11-08T13:10:18.973155Z","shell.execute_reply.started":"2025-11-08T13:10:18.966825Z","shell.execute_reply":"2025-11-08T13:10:18.971545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\ndef plot_simple_loss_graph(history):\n    \"\"\"Generates a simple line plot of the provided loss history.\"\"\"\n    epochs = range(1, len(history) + 1)\n    \n    plt.figure(figsize=(8, 5))\n    \n    # Plotting the loss\n    plt.plot(epochs, history, label='Training Loss', marker='o', linestyle='-', color='blue')\n    \n    plt.title('Model Training Progress (Loss vs. Epochs)')\n    plt.xlabel('Epoch Number')\n    plt.ylabel('Loss Value')\n    plt.legend()\n    plt.grid(True)\n    plt.show()\n\n# --- Example Usage: ---\n\n# This is the dummy data you would typically collect during a training run:\nsample_loss_history = [0.8, 0.6, 0.4, 0.3, 0.25, 0.21, 0.18, 0.16, 0.15, 0.14]\n\n# Call the function to display the plot\nplot_simple_loss_graph(sample_loss_history)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:12:25.592400Z","iopub.execute_input":"2025-11-08T13:12:25.592756Z","iopub.status.idle":"2025-11-08T13:12:25.804226Z","shell.execute_reply.started":"2025-11-08T13:12:25.592723Z","shell.execute_reply":"2025-11-08T13:12:25.802759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_mask = np.array([[1, 1, 0, 0, 1, 0]])\nprint(f\"Test mask: {test_mask.flatten()}\")\n\n# Step by step explanation\npixels = test_mask.flatten()\nprint(f\"1. Flatten: {pixels}\")\n\npadded = np.concatenate([[0], pixels, [0]])\nprint(f\"2. Add borders: {padded}\")\n\nchanges = np.where(padded[1:] != padded[:-1])[0] + 1\nprint(f\"3. Find changes: {changes}\")\n\nruns = changes.copy()\nruns[1::2] -= runs[::2]\nprint(f\"4. Calculate lengths: {runs}\")\n\nresult = ' '.join(str(x) for x in runs)\nprint(f\"5. Final RLE: '{result}'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:10:22.058167Z","iopub.execute_input":"2025-11-08T13:10:22.058518Z","iopub.status.idle":"2025-11-08T13:10:22.068662Z","shell.execute_reply.started":"2025-11-08T13:10:22.058493Z","shell.execute_reply":"2025-11-08T13:10:22.067218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\n\n# 1. Prepare dummy data (replace with your actual data DataFrame)\n# Example of metadata you might collect about your images\ndata = {\n    'Image_Width': np.random.randint(100, 500, 50),\n    'Image_Height': np.random.randint(100, 500, 50),\n    'Pixel_Mean': np.random.rand(50) * 0.5 + 0.25,\n    'Forgery_Area_Ratio': np.random.rand(50) * 0.2\n}\ndf = pd.DataFrame(data)\ndf['Aspect_Ratio'] = df['Image_Width'] / df['Image_Height']\n\n# 2. Calculate the correlation matrix\ncorr_matrix = df.corr()\n\n# 3. Create the heatmap plot\nplt.figure(figsize=(8, 6))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', fmt=\".2f\", linewidths=.5)\n\nplt.title('Feature Correlation Heatmap')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:13:26.585805Z","iopub.execute_input":"2025-11-08T13:13:26.586797Z","iopub.status.idle":"2025-11-08T13:13:26.912986Z","shell.execute_reply.started":"2025-11-08T13:13:26.586758Z","shell.execute_reply":"2025-11-08T13:13:26.911668Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\n\n# 1. Prepare dummy data with categories\n# Data for 'authentic' and 'forged' categories\nauthentic_areas = np.random.rand(50) * 10\nforged_areas = np.random.rand(50) * 50 # Forged images might have larger altered areas\n\ndata_to_plot = [authentic_areas, forged_areas]\nlabels = ['Authentic', 'Forged']\n\n# 2. Create the box plot\nplt.figure(figsize=(7, 5))\nplt.boxplot(data_to_plot, labels=labels, patch_artist=True, vert=True)\n\n# 3. Add labels and title\nplt.title('Distribution of Forgery Areas by Category')\nplt.ylabel('Area Size (pixels/ratio)')\nplt.xlabel('Image Type')\nplt.grid(axis='y')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:14:10.692362Z","iopub.execute_input":"2025-11-08T13:14:10.692694Z","iopub.status.idle":"2025-11-08T13:14:10.862904Z","shell.execute_reply.started":"2025-11-08T13:14:10.692672Z","shell.execute_reply":"2025-11-08T13:14:10.861558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# 1. Prepare dummy data: 50 authentic, 50 forged (100 total)\nnp.random.seed(42)\nimage_widths = np.random.randint(100, 500, 100)\nforgery_ratios = np.random.rand(100) * 0.2\n# Create a color map array: 0 for Authentic (first 50), 1 for Forged (last 50)\ncategories = np.array([0] * 50 + [1] * 50) \n\nplt.figure(figsize=(8, 6))\n\n# Use a colormap ('viridis') and the 'c' argument to map colors to the 'categories' array\nscatter = plt.scatter(image_widths, forgery_ratios, c=categories, cmap='viridis', alpha=0.7)\n\n# Create a legend manually (more complex in pure matplotlib than in seaborn/plotly)\nhandles, labels = scatter.legend_elements(prop=\"colors\", alpha=0.6)\nlegend_labels = ['Authentic', 'Forged']\nplt.legend(handles, legend_labels, loc=\"lower right\", title=\"Type\")\n\nplt.xlabel('Image Width')\nplt.ylabel('Forgery Area Ratio')\nplt.title('Scatter Plot colored by Category (Matplotlib)')\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:15:14.538051Z","iopub.execute_input":"2025-11-08T13:15:14.538367Z","iopub.status.idle":"2025-11-08T13:15:14.810517Z","shell.execute_reply.started":"2025-11-08T13:15:14.538347Z","shell.execute_reply":"2025-11-08T13:15:14.809059Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# 1. Prepare dummy data: Counts of images that have a forgery vs those that don't, \n# categorized by some \"source\" or \"folder A/B\"\ncategories = ['Source A', 'Source B', 'Source C']\nhas_forgery = np.array([200, 150, 300])\nno_forgery = np.array([300, 350, 200]) # Ensures totals are 500 each\n\n# 2. Plotting the stacked bars\nfig, ax = plt.subplots(figsize=(8, 5))\n\n# Plot 'no_forgery' bars first (bottom layer)\nax.bar(categories, no_forgery, label='No Forgery (Authentic)', color='skyblue')\n\n# Plot 'has_forgery' bars on top (using 'bottom=no_forgery' to stack them)\nax.bar(categories, has_forgery, bottom=no_forgery, label='Has Forgery (Forged)', color='coral')\n\nax.set_ylabel('Number of Images')\nax.set_title('Image Counts by Source and Forgery Status')\nax.legend()\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:15:32.610417Z","iopub.execute_input":"2025-11-08T13:15:32.610784Z","iopub.status.idle":"2025-11-08T13:15:32.797990Z","shell.execute_reply.started":"2025-11-08T13:15:32.610760Z","shell.execute_reply":"2025-11-08T13:15:32.796784Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport numpy as np\n\n# 1. Prepare dummy data (100 images)\nwidths = np.random.randint(100, 500, 100)\nheights = np.random.randint(100, 500, 100)\n# 'File size' in MB for each image (used to determine marker size)\nfile_sizes_mb = np.random.rand(100) * 5 + 1 \n\n# Scale file size so points aren't too small/large on the plot\nmarker_sizes = file_sizes_mb * 50\n\nplt.figure(figsize=(8, 6))\n\n# Use the 's' parameter for marker size\nscatter = plt.scatter(widths, heights, s=marker_sizes, alpha=0.5, color='purple', edgecolors='black')\n\nplt.xlabel('Image Width')\nplt.ylabel('Image Height')\nplt.title('Image Dimensions Colored by File Size')\nplt.grid(True)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T13:16:11.138420Z","iopub.execute_input":"2025-11-08T13:16:11.138779Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div style=\"\n    background: linear-gradient(135deg, #1a1f2c 0%, #2d3748 50%, #4a5568 100%);\n    border: 2px solid #63b3ed;\n    border-radius: 15px;\n    padding: 25px;\n    margin: 20px 0;\n    box-shadow: 0 0 30px rgba(99, 179, 237, 0.4),\n                inset 0 0 20px rgba(255, 255, 255, 0.1);\n    color: #f1f5f9;\n    font-family: 'Segoe UI', system-ui, sans-serif;\n    position: relative;\n    overflow: hidden;\n\">\n\n<div style=\"\n    position: absolute;\n    top: -20px;\n    right: -20px;\n    width: 100px;\n    height: 100px;\n    background: radial-gradient(circle, rgba(99, 179, 237, 0.25) 0%, transparent 70%);\n    border-radius: 50%;\n\"></div>\n\n<div style=\"\n    position: absolute;\n    bottom: -40px;\n    left: -40px;\n    width: 120px;\n    height: 120px;\n    background: radial-gradient(circle, rgba(99, 179, 237, 0.2) 0%, transparent 70%);\n    border-radius: 50%;\n\"></div>\n\n<h1 style=\"\n    color: #63b3ed;\n    margin: 0 0 20px 0;\n    text-align: center;\n    font-weight: 700;\n    font-size: 1.8em;\n    text-shadow: 0 0 15px rgba(99, 179, 237, 0.6);\n    position: relative;\n    z-index: 1;\n\">\n    Create sumission distributed by the most frequent position in the mask\n</h1>","metadata":{}},{"cell_type":"code","source":"import os\nimport cv2\nimport json\nimport numpy as np\nimport pandas as pd\n\nfrom PIL import Image\nfrom scipy.stats import gaussian_kde\n\nnp.random.seed(81)\n\nsample_submission_path = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/sample_submission.csv'\ntest_images_dir = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/test_images'\ntrain_masks_dir = '/kaggle/input/recodai-luc-scientific-image-forgery-detection/train_masks'\n\nsample_submission = pd.read_csv(sample_submission_path)\n\ndef analyze_mask_distribution():\n    if not os.path.exists(train_masks_dir):\n        return None, None, None\n    \n    all_positions = []\n    all_sizes = []\n    all_aspect_ratios = []\n    \n    for mask_file in os.listdir(train_masks_dir):\n        if mask_file.endswith('.npy'):\n            mask_path = os.path.join(train_masks_dir, mask_file)\n            try:\n                mask = np.load(mask_path)\n                \n                if mask.ndim == 3:\n                    if mask.shape[0] == 1:\n                        mask = mask[0]\n                    elif mask.shape[2] == 1:\n                        mask = mask[:, :, 0]\n                    else:\n                        mask = (mask > 0).astype(np.uint8)\n                        if mask.ndim == 3:\n                            mask = mask[:, :, 0] if mask.shape[2] == 1 else mask[:, :, 0]\n                \n                if mask.ndim != 2 or np.sum(mask) == 0:\n                    continue\n                \n                contours, _ = cv2.findContours(mask.astype(np.uint8), cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n                \n                for contour in contours:\n                    if len(contour) >= 5:\n                        ellipse = cv2.fitEllipse(contour)\n                        center, axes, angle = ellipse\n                        \n                        height, width = mask.shape\n                        norm_center_x = center[0] / width\n                        norm_center_y = center[1] / height\n                        \n                        norm_major_axis = max(axes) / max(width, height)\n                        norm_minor_axis = min(axes) / max(width, height)\n                        \n                        aspect_ratio = min(axes) / max(axes) if max(axes) > 0 else 1.0\n                        \n                        all_positions.append((norm_center_x, norm_center_y))\n                        all_sizes.append((norm_major_axis, norm_minor_axis))\n                        all_aspect_ratios.append(aspect_ratio)\n                        \n            except Exception:\n                continue\n    \n    if not all_positions:\n        return None, None, None\n    \n    positions_array = np.array(all_positions)\n    if len(positions_array) > 1:\n        kde = gaussian_kde(positions_array.T)\n    else:\n        kde = None\n    \n    sizes_array = np.array(all_sizes)\n    aspect_ratios_array = np.array(all_aspect_ratios)\n    \n    size_mean = np.mean(sizes_array, axis=0) if len(sizes_array) > 0 else (0.02, 0.01)\n    size_std = np.std(sizes_array, axis=0) if len(sizes_array) > 0 else (0.01, 0.005)\n    aspect_mean = np.mean(aspect_ratios_array) if len(aspect_ratios_array) > 0 else 0.7\n    aspect_std = np.std(aspect_ratios_array) if len(aspect_ratios_array) > 0 else 0.2\n    \n    return kde, (size_mean, size_std), (aspect_mean, aspect_std)\n\ndef generate_ellipse_mask(height, width, kde, size_stats, aspect_stats):\n    size_mean, size_std = size_stats\n    aspect_mean, aspect_std = aspect_stats\n    \n    if kde and len(kde.dataset.T) > 1:\n        random_idx = np.random.randint(0, len(kde.dataset.T))\n        center_x, center_y = kde.dataset.T[random_idx]\n    else:\n        center_x = np.random.normal(0.5, 0.2)\n        center_y = np.random.normal(0.5, 0.2)\n        center_x = np.clip(center_x, 0.1, 0.9)\n        center_y = np.clip(center_y, 0.1, 0.9)\n    \n    major_axis = np.random.normal(size_mean[0], size_std[0])\n    minor_axis = np.random.normal(size_mean[1], size_std[1])\n    \n    major_axis = np.clip(major_axis, 0.005, 0.1)\n    minor_axis = np.clip(minor_axis, 0.003, 0.08)\n    \n    aspect_ratio = np.random.normal(aspect_mean, aspect_std)\n    aspect_ratio = np.clip(aspect_ratio, 0.3, 0.95)\n    minor_axis = major_axis * aspect_ratio\n    \n    mask = np.zeros((height, width), dtype=np.uint8)\n    \n    abs_center_x = int(center_x * width)\n    abs_center_y = int(center_y * height)\n    abs_major = int(major_axis * max(height, width))\n    abs_minor = int(minor_axis * max(height, width))\n    \n    abs_major = max(abs_major, 2)\n    abs_minor = max(abs_minor, 2)\n    \n    cv2.ellipse(mask, \n                (abs_center_x, abs_center_y),\n                (abs_minor, abs_major),\n                angle=np.random.uniform(0, 180),\n                startAngle=0,\n                endAngle=360,\n                color=1,\n                thickness=-1)\n    \n    if np.random.random() < 0.3:\n        kernel_size = np.random.choice([1, 3])\n        if kernel_size > 1:\n            mask = cv2.GaussianBlur(mask.astype(np.float32), (kernel_size, kernel_size), 0)\n            mask = (mask > 0.3).astype(np.uint8)\n    \n    return mask\n\ndef generate_irregular_mask(height, width, kde, size_stats):\n    mask = generate_ellipse_mask(height, width, kde, size_stats, (0.7, 0.2))\n    \n    if np.random.random() < 0.5:\n        kernel = np.ones((2, 2), np.uint8)\n        if np.random.random() < 0.5:\n            mask = cv2.erode(mask, kernel, iterations=1)\n        else:\n            mask = cv2.dilate(mask, kernel, iterations=1)\n    \n    return mask\n\ndef generate_multiple_masks(height, width, kde, size_stats, aspect_stats):\n    num_masks = np.random.choice([1, 2, 3], p=[0.7, 0.2, 0.1])\n    final_mask = np.zeros((height, width), dtype=np.uint8)\n    \n    for _ in range(num_masks):\n        mask = generate_ellipse_mask(height, width, kde, size_stats, aspect_stats)\n        final_mask = np.logical_or(final_mask, mask)\n    \n    return final_mask.astype(np.uint8)\n\ndef rle_encode(mask):\n    pixels = mask.flatten()\n    pixels = np.concatenate([[0], pixels, [0]])\n    runs = np.where(pixels[1:] != pixels[:-1])[0] + 1\n    runs[1::2] -= runs[::2]\n    return ' '.join(str(x) for x in runs)\n\nkde, size_stats, aspect_stats = analyze_mask_distribution()\n\nsubmission_data = []\n\nfor case_id in sample_submission['case_id']:\n    img_path = os.path.join(test_images_dir, f\"{case_id}.png\")\n    \n    if not os.path.exists(img_path):\n        annotation = 'authentic'\n    else:\n        with Image.open(img_path) as img:\n            width, height = img.size\n        \n        if np.random.random() < 0.005:\n            mask_type = np.random.choice(['ellipse', 'irregular', 'multiple'], p=[0.6, 0.3, 0.1])\n            \n            if mask_type == 'ellipse':\n                mask = generate_ellipse_mask(height, width, kde, size_stats, aspect_stats)\n            elif mask_type == 'irregular':\n                mask = generate_irregular_mask(height, width, kde, size_stats)\n            else:\n                mask = generate_multiple_masks(height, width, kde, size_stats, aspect_stats)\n            \n            if np.sum(mask) > 0:\n                RLE_res = rle_encode(mask)\n                res = [int(x) for x in RLE_res.split()]\n                annotation = json.dumps(res)\n            else:\n                annotation = 'authentic'\n        else:\n            annotation = 'authentic'\n    \n    submission_data.append({\n        'case_id': case_id,\n        'annotation': annotation\n    })\n\nsubmission = pd.DataFrame(submission_data)\nsubmission.to_csv('submission.csv', index=False)\nsubmission.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-08T12:38:38.605490Z","iopub.execute_input":"2025-11-08T12:38:38.605829Z","iopub.status.idle":"2025-11-08T12:38:49.575669Z","shell.execute_reply.started":"2025-11-08T12:38:38.605806Z","shell.execute_reply":"2025-11-08T12:38:49.574392Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}