{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":113558,"databundleVersionId":14174843,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from pathlib import Path\nimport pandas as pd\n\nPATH_DATASET = \"/kaggle/input/recodai-luc-scientific-image-forgery-detection\"\nTEST_IMAGES_DIR = f'{PATH_DATASET}/test_images'\ntest_img_dir = Path(TEST_IMAGES_DIR) # Need test_img_dir to get all image names\n\n# Get a list of all test image filenames (without extension)\ntest_image_stems = sorted([img_path.stem for img_path in test_img_dir.glob('*.png')])\nprint(f\"Found images: {len(test_image_stems)}\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-06T10:33:40.970382Z","iopub.execute_input":"2025-11-06T10:33:40.970933Z","iopub.status.idle":"2025-11-06T10:33:40.977328Z","shell.execute_reply.started":"2025-11-06T10:33:40.970910Z","shell.execute_reply":"2025-11-06T10:33:40.976357Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def rle_encode(mask, fg_val=1):\n    \"\"\"Convert binary mask to RLE using the competition metric format\"\"\"\n    dots = np.where(mask.T.flatten() == fg_val)[0]\n    run_lengths = []\n    prev = -2\n    \n    for b in dots:\n        if b > prev + 1:\n            run_lengths.extend((b + 1, 0))\n        run_lengths[-1] += 1\n        prev = b\n    \n    return run_lengths","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T10:33:40.978692Z","iopub.execute_input":"2025-11-06T10:33:40.979227Z","iopub.status.idle":"2025-11-06T10:33:41.000297Z","shell.execute_reply.started":"2025-11-06T10:33:40.979211Z","shell.execute_reply":"2025-11-06T10:33:40.999475Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nfrom tqdm.auto import tqdm\n\nsubmissions = []\n\nfor img_stem in tqdm(test_image_stems, desc=\"Creating submission\"):\n    # mask_path = masks_dir / f\"{img_stem}.npy\"\n    # mask = np.load(mask_path)\n\n    mask = np.zeros((50, 50), dtype=int)\n    # Randomly choose to generate a mask\n    if np.random.rand() > 0.5:\n        mask[10:15, 20:30] = 1\n\n    rle_annotation = rle_encode(mask)\n    print(rle_annotation)\n    if rle_annotation: # If any pixels are marked as forgery\n        submissions.append({'case_id': img_stem, 'annotation': f'\"{repr(rle_annotation)}\"'})\n    else: # For authentic images, the annotation should be 'authentic' without quotes\n        submissions.append({'case_id': img_stem, 'annotation': 'authentic'})\n\n\nsubmission_df = pd.DataFrame(submissions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T10:34:19.606771Z","iopub.execute_input":"2025-11-06T10:34:19.607123Z","iopub.status.idle":"2025-11-06T10:34:19.627316Z","shell.execute_reply.started":"2025-11-06T10:34:19.607101Z","shell.execute_reply":"2025-11-06T10:34:19.625755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pprint import pprint\n\ndef write_submission_csv(data, filename):\n    \"\"\"Writes submission data to a CSV file using writelines by generating lines upfront.\"\"\"\n    lines = [\"case_id,annotation\\n\"]  # Header line\n    for row in data:\n        # Ensure the annotation is correctly formatted with quotes if it's an RLE string\n        annotation = row['annotation']\n        lines.append(f\"{row['case_id']},{annotation}\\n\")\n    pprint(lines)\n    with open(filename, 'w') as f:\n        f.writelines(lines)\n\n\n# Assuming 'submissions' list is already created from the previous inference step\n# (or you can re-run the inference part to generate it)\nwrite_submission_csv(submissions, 'submission.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T10:34:42.308750Z","iopub.execute_input":"2025-11-06T10:34:42.309042Z","iopub.status.idle":"2025-11-06T10:34:42.315523Z","shell.execute_reply.started":"2025-11-06T10:34:42.309028Z","shell.execute_reply":"2025-11-06T10:34:42.314667Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!head submission.csv","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-06T10:34:44.338838Z","iopub.execute_input":"2025-11-06T10:34:44.339129Z","iopub.status.idle":"2025-11-06T10:34:44.463180Z","shell.execute_reply.started":"2025-11-06T10:34:44.339112Z","shell.execute_reply":"2025-11-06T10:34:44.462062Z"}},"outputs":[],"execution_count":null}]}