{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":113558,"databundleVersionId":14456136,"sourceType":"competition"}],"dockerImageVersionId":31192,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import json\nimport os\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport cv2\nimport numba\nimport numpy as np\nfrom numba import types\nimport numpy.typing as npt\nimport pandas as pd\nimport scipy.optimize\n\n\nclass ParticipantVisibleError(Exception):\n    pass\n\n\n@numba.jit(nopython=True)\ndef _rle_encode_jit(x: npt.NDArray, fg_val: int = 1) -> list[int]:\n    \"\"\"Numba-jitted RLE encoder.\"\"\"\n    dots = np.where(x.T.flatten() == fg_val)[0]\n    run_lengths = []\n    prev = -2\n    for b in dots:\n        if b > prev + 1:\n            run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    return run_lengths\n\n\ndef rle_encode(masks: list[npt.NDArray], fg_val: int = 1) -> str:\n    \"\"\"\n    Adapted from contrails RLE https://www.kaggle.com/code/inversion/contrails-rle-submission\n    Args:\n        masks: list of numpy array of shape (height, width), 1 - mask, 0 - background\n    Returns: run length encodings as a string, with each RLE JSON-encoded and separated by a semicolon.\n    \"\"\"\n    return ';'.join([json.dumps(_rle_encode_jit(x, fg_val)) for x in masks])\n\n\n@numba.njit\ndef _rle_decode_jit(mask_rle: npt.NDArray, height: int, width: int) -> npt.NDArray:\n    \"\"\"\n    s: numpy array of run-length encoding pairs (start, length)\n    shape: (height, width) of array to return\n    Returns numpy array, 1 - mask, 0 - background\n    \"\"\"\n    if len(mask_rle) % 2 != 0:\n        # Numba requires raising a standard exception.\n        raise ValueError('One or more rows has an odd number of values.')\n\n    starts, lengths = mask_rle[0::2], mask_rle[1::2]\n    starts -= 1\n    ends = starts + lengths\n    for i in range(len(starts) - 1):\n        if ends[i] > starts[i + 1]:\n            raise ValueError('Pixels must not be overlapping.')\n    img = np.zeros(height * width, dtype=np.bool_)\n    for lo, hi in zip(starts, ends):\n        img[lo:hi] = 1\n    return img\n\n\ndef rle_decode(mask_rle: str, shape: tuple[int, int]) -> npt.NDArray:\n    \"\"\"\n    mask_rle: run-length as string formatted (start length)\n              empty predictions need to be encoded with '-'\n    shape: (height, width) of array to return\n    Returns numpy array, 1 - mask, 0 - background\n    \"\"\"\n\n    mask_rle = json.loads(mask_rle)\n    mask_rle = np.asarray(mask_rle, dtype=np.int32)\n    starts = mask_rle[0::2]\n    if sorted(starts) != list(starts):\n        raise ParticipantVisibleError('Submitted values must be in ascending order.')\n    try:\n        return _rle_decode_jit(mask_rle, shape[0], shape[1]).reshape(shape, order='F')\n    except ValueError as e:\n        raise ParticipantVisibleError(str(e)) from e\n\n\ndef calculate_f1_score(pred_mask: npt.NDArray, gt_mask: npt.NDArray):\n    pred_flat = pred_mask.flatten()\n    gt_flat = gt_mask.flatten()\n\n    tp = np.sum((pred_flat == 1) & (gt_flat == 1))\n    fp = np.sum((pred_flat == 1) & (gt_flat == 0))\n    fn = np.sum((pred_flat == 0) & (gt_flat == 1))\n\n    precision = tp / (tp + fp) if (tp + fp) > 0 else 0\n    recall = tp / (tp + fn) if (tp + fn) > 0 else 0\n\n    if (precision + recall) > 0:\n        return 2 * (precision * recall) / (precision + recall)\n    else:\n        return 0\n\n\ndef calculate_f1_matrix(pred_masks: list[npt.NDArray], gt_masks: list[npt.NDArray]):\n    \"\"\"\n    Parameters:\n    pred_masks (np.ndarray):\n            First dimension is the number of predicted instances.\n            Each instance is a binary mask of shape (height, width).\n    gt_masks (np.ndarray):\n            First dimension is the number of ground truth instances.\n            Each instance is a binary mask of shape (height, width).\n    \"\"\"\n\n    num_instances_pred = len(pred_masks)\n    num_instances_gt = len(gt_masks)\n    f1_matrix = np.zeros((num_instances_pred, num_instances_gt))\n\n    # Calculate F1 scores for each pair of predicted and ground truth masks\n    for i in range(num_instances_pred):\n        for j in range(num_instances_gt):\n            pred_flat = pred_masks[i].flatten()\n            gt_flat = gt_masks[j].flatten()\n            f1_matrix[i, j] = calculate_f1_score(pred_mask=pred_flat, gt_mask=gt_flat)\n\n    if f1_matrix.shape[0] < len(gt_masks):\n        # Add a row of zeros to the matrix if the number of predicted instances is less than ground truth instances\n        f1_matrix = np.vstack((f1_matrix, np.zeros((len(gt_masks) - len(f1_matrix), num_instances_gt))))\n\n    return f1_matrix\n\n\ndef oF1_score(pred_masks: list[npt.NDArray], gt_masks: list[npt.NDArray]):\n    \"\"\"\n    Calculate the optimal F1 score for a set of predicted masks against\n    ground truth masks which considers the optimal F1 score matching.\n    This function uses the Hungarian algorithm to find the optimal assignment\n    of predicted masks to ground truth masks based on the F1 score matrix.\n    If the number of predicted masks is less than the number of ground truth masks,\n    it will add a row of zeros to the F1 score matrix to ensure that the dimensions match.\n\n    Parameters:\n    pred_masks (list of np.ndarray): List of predicted binary masks.\n    gt_masks (np.ndarray): Array of ground truth binary masks.\n    Returns:\n    float: Optimal F1 score.\n    \"\"\"\n    f1_matrix = calculate_f1_matrix(pred_masks, gt_masks)\n\n    # Find the best matching between predicted and ground truth masks\n    row_ind, col_ind = scipy.optimize.linear_sum_assignment(-f1_matrix)\n    # The linear_sum_assignment discards excess predictions so we need a separate penalty.\n    excess_predictions_penalty = len(gt_masks) / max(len(pred_masks), len(gt_masks))\n    return np.mean(f1_matrix[row_ind, col_ind]) * excess_predictions_penalty\n\n\ndef evaluate_single_image(label_rles: str, prediction_rles: str, shape_str: str) -> float:\n    shape = json.loads(shape_str)\n    label_rles = [rle_decode(x, shape=shape) for x in label_rles.split(';')]\n    prediction_rles = [rle_decode(x, shape=shape) for x in prediction_rles.split(';')]\n    return oF1_score(prediction_rles, label_rles)\n\n\ndef score(solution: pd.DataFrame, submission: pd.DataFrame, row_id_column_name: str) -> float:\n    \"\"\"\n    Args:\n        solution (pd.DataFrame): The ground truth DataFrame.\n        submission (pd.DataFrame): The submission DataFrame.\n        row_id_column_name (str): The name of the column containing row IDs.\n    Returns:\n        float\n\n    Examples\n    --------\n    >>> solution = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['authentic', 'authentic', 'authentic'], 'shape': ['authentic', 'authentic', 'authentic']})\n    >>> submission = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['authentic', 'authentic', 'authentic']})\n    >>> score(solution.copy(), submission.copy(), row_id_column_name='row_id')\n    1.0\n\n    >>> solution = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['authentic', 'authentic', 'authentic'], 'shape': ['authentic', 'authentic', 'authentic']})\n    >>> submission = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 102]', '[101, 102]', '[101, 102]']})\n    >>> score(solution.copy(), submission.copy(), row_id_column_name='row_id')\n    0.0\n\n    >>> solution = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 102]', '[101, 102]', '[101, 102]'], 'shape': ['[720, 960]', '[720, 960]', '[720, 960]']})\n    >>> submission = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 102]', '[101, 102]', '[101, 102]']})\n    >>> score(solution.copy(), submission.copy(), row_id_column_name='row_id')\n    1.0\n\n    >>> solution = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 103]', '[101, 102]', '[101, 102]'], 'shape': ['[720, 960]', '[720, 960]', '[720, 960]']})\n    >>> submission = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 102]', '[101, 102]', '[101, 102]']})\n    >>> score(solution.copy(), submission.copy(), row_id_column_name='row_id')\n    0.9983739837398374\n\n    >>> solution = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 102];[300, 100]', '[101, 102]', '[101, 102]'], 'shape': ['[720, 960]', '[720, 960]', '[720, 960]']})\n    >>> submission = pd.DataFrame({'row_id': [0, 1, 2], 'annotation': ['[101, 102]', '[101, 102]', '[101, 102]']})\n    >>> score(solution.copy(), submission.copy(), row_id_column_name='row_id')\n    0.8333333333333334\n    \"\"\"\n    df = solution\n    df = df.rename(columns={'annotation': 'label'})\n\n    df['prediction'] = submission['annotation']\n    # Check for correct 'authentic' label\n    authentic_indices = (df['label'] == 'authentic') | (df['prediction'] == 'authentic')\n    df['image_score'] = ((df['label'] == df['prediction']) & authentic_indices).astype(float)\n\n    df.loc[~authentic_indices, 'image_score'] = df.loc[~authentic_indices].apply(\n        lambda row: evaluate_single_image(row['label'], row['prediction'], row['shape']), axis=1\n    )\n    return float(np.mean(df['image_score']))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-26T15:27:46.914438Z","iopub.execute_input":"2025-11-26T15:27:46.914750Z","iopub.status.idle":"2025-11-26T15:27:51.032775Z","shell.execute_reply.started":"2025-11-26T15:27:46.914720Z","shell.execute_reply":"2025-11-26T15:27:51.031791Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"f_images = os.listdir('/kaggle/input/recodai-luc-scientific-image-forgery-detection/train_images/forged')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T15:27:51.034575Z","iopub.execute_input":"2025-11-26T15:27:51.035142Z","iopub.status.idle":"2025-11-26T15:27:51.177099Z","shell.execute_reply.started":"2025-11-26T15:27:51.035107Z","shell.execute_reply":"2025-11-26T15:27:51.175988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for i in range(len(f_images)):\n    forg = Image.open('/kaggle/input/recodai-luc-scientific-image-forgery-detection/train_images/forged/'+f_images[i])\n    f_img = np.array(forg.convert('RGB'))\n    mask = np.load('/kaggle/input/recodai-luc-scientific-image-forgery-detection/train_masks/'+f_images[i][:-3]+'npy')\n    print('Image')\n    plt.imshow(f_img)\n    plt.show()\n    for m in range(np.shape(mask)[0]):\n        print('Mask_Instance:', m)\n        plt.imshow(mask[m])\n        plt.show()\n    break","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T15:27:51.178146Z","iopub.execute_input":"2025-11-26T15:27:51.178494Z","iopub.status.idle":"2025-11-26T15:27:51.881322Z","shell.execute_reply.started":"2025-11-26T15:27:51.178462Z","shell.execute_reply":"2025-11-26T15:27:51.880023Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Test Evaluation with ground_truth:","metadata":{}},{"cell_type":"code","source":"evaluate_single_image(rle_encode(mask), rle_encode(mask), f'[{np.shape(mask)[1]}, {np.shape(mask)[2]}]')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T15:30:22.114561Z","iopub.execute_input":"2025-11-26T15:30:22.115143Z","iopub.status.idle":"2025-11-26T15:30:24.646406Z","shell.execute_reply.started":"2025-11-26T15:30:22.115034Z","shell.execute_reply":"2025-11-26T15:30:24.645554Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Okay, f1_score is 1.0 for a perfect prediction.\n\nBut when your model predicts all masks as ONE instance:","metadata":{}},{"cell_type":"code","source":"melted = mask[0] + mask[1]\nplt.imshow(melted)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T15:36:06.326156Z","iopub.execute_input":"2025-11-26T15:36:06.326429Z","iopub.status.idle":"2025-11-26T15:36:06.490294Z","shell.execute_reply.started":"2025-11-26T15:36:06.326411Z","shell.execute_reply":"2025-11-26T15:36:06.489240Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"nearly perfect prediction; all masks are found. But what happens with the competition metric?","metadata":{}},{"cell_type":"code","source":"evaluate_single_image(rle_encode(mask), rle_encode(melted), f'[{np.shape(mask)[1]}, {np.shape(mask)[2]}]')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T15:34:18.119335Z","iopub.execute_input":"2025-11-26T15:34:18.119601Z","iopub.status.idle":"2025-11-26T15:34:22.306262Z","shell.execute_reply.started":"2025-11-26T15:34:18.119581Z","shell.execute_reply":"2025-11-26T15:34:22.305285Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The f1-score decreases to 0 !!! But our new 'melted' mask has now a dimension of (512, 696), and the ground_truth (2, 512, 696). So we try a new approach with added dimension 0:","metadata":{}},{"cell_type":"code","source":"melted = np.expand_dims(melted, axis = 0)\nevaluate_single_image(rle_encode(mask), rle_encode(melted), f'[{np.shape(mask)[1]}, {np.shape(mask)[2]}]')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-26T15:39:11.437332Z","iopub.execute_input":"2025-11-26T15:39:11.438171Z","iopub.status.idle":"2025-11-26T15:39:11.455335Z","shell.execute_reply.started":"2025-11-26T15:39:11.438143Z","shell.execute_reply":"2025-11-26T15:39:11.454494Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"After adding the dimension, now the f1-score is 34.82%. Once again, all masks predicted perfect, but now only in one instance instead of two. Is this penalization fair?","metadata":{}}]}