{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":113558,"databundleVersionId":14456136,"sourceType":"competition"}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-11-25T04:52:13.672939Z","iopub.execute_input":"2025-11-25T04:52:13.673165Z","iopub.status.idle":"2025-11-25T04:52:25.9749Z","shell.execute_reply.started":"2025-11-25T04:52:13.67314Z","shell.execute_reply":"2025-11-25T04:52:25.974265Z"},"_kg_hide-output":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport random\nfrom glob import glob\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\n\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import layers\n\n\n# Reproducibility\nSEED = 42\nrandom.seed(SEED)\nnp.random.seed(SEED)\ntf.random.set_seed(SEED)\n\n\n# =====================\n# 2) Paths & Overview\n# =====================\nBASE = \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/supplemental_images\" # change if needed\nTRAIN_AUTH = os.path.join(BASE, \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/train_images/authentic\")\nTRAIN_FORG = os.path.join(BASE, \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/train_images/forged\")\nTRAIN_MASKS = os.path.join(BASE, \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/train_images/forged\")\nSUPP_IMG = os.path.join(BASE, \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/supplemental_images\")\nSUPP_MASK = os.path.join(BASE, \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/supplemental_masks\")\nTEST_DIR = os.path.join(BASE, \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/test_images\")\nSAMPLE_SUB = os.path.join(BASE, \"/kaggle/input/recodai-luc-scientific-image-forgery-detection/sample_submission.csv\")\n\n\nprint(\"Paths set. Listing counts:\")\nprint(\"Authentic:\", len(glob(os.path.join(TRAIN_AUTH, \"*.png\"))))\nprint(\"Forged: \", len(glob(os.path.join(TRAIN_FORG, \"*.png\"))))\nprint(\"Masks: \", len(glob(os.path.join(TRAIN_MASKS, \"*.png\"))))\nprint(\"Test: \", len(glob(os.path.join(TEST_DIR, \"*.png\"))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T05:02:18.994587Z","iopub.execute_input":"2025-11-25T05:02:18.995141Z","iopub.status.idle":"2025-11-25T05:02:19.021759Z","shell.execute_reply.started":"2025-11-25T05:02:18.995109Z","shell.execute_reply":"2025-11-25T05:02:19.020982Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 3) Helper functions\n# =====================\nimport cv2\n\ndef read_image(path, size=(256,256), grayscale=False):\n    img = cv2.imread(path, cv2.IMREAD_COLOR if not grayscale else cv2.IMREAD_GRAYSCALE)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB) if not grayscale else img\n    img = cv2.resize(img, size, interpolation=cv2.INTER_AREA)\n    img = img.astype(np.float32) / 255.0\n    return img\n\n\ndef read_mask(path, size=(256,256)):\n    m = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n    m = cv2.resize(m, size, interpolation=cv2.INTER_NEAREST)\n    m = (m > 127).astype(np.float32)\n    m = np.expand_dims(m, -1)\n    return m","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T05:03:09.433084Z","iopub.execute_input":"2025-11-25T05:03:09.433385Z","iopub.status.idle":"2025-11-25T05:03:09.438934Z","shell.execute_reply.started":"2025-11-25T05:03:09.433363Z","shell.execute_reply":"2025-11-25T05:03:09.438258Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Utility to show images\ndef show_images_grid(images, titles=None, cols=4, figsize=(12,8)):\n    rows = (len(images) + cols - 1) // cols\n    plt.figure(figsize=figsize)\n    for i, img in enumerate(images):\n        plt.subplot(rows, cols, i+1)\n        plt.imshow(img if img.ndim==3 else img.squeeze(), cmap='gray')\n        if titles: plt.title(titles[i])\n        plt.axis('off')\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T05:03:31.062477Z","iopub.execute_input":"2025-11-25T05:03:31.06277Z","iopub.status.idle":"2025-11-25T05:03:31.067487Z","shell.execute_reply.started":"2025-11-25T05:03:31.062749Z","shell.execute_reply":"2025-11-25T05:03:31.066831Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# 4) Data listing & EDA\n# =====================\nauth_paths = sorted(glob(os.path.join(TRAIN_AUTH, \"*.png\")))\nforg_paths = sorted(glob(os.path.join(TRAIN_FORG, \"*.png\")))\nmask_paths = sorted(glob(os.path.join(TRAIN_MASKS, \"*.png\")))\n\nprint(f\"auth: {len(auth_paths)} | forg: {len(forg_paths)} | masks: {len(mask_paths)}\")\n\n# show samples\nsample_imgs = []\nif len(auth_paths)>0: sample_imgs.append(read_image(auth_paths[0]))\nif len(forg_paths)>0: sample_imgs.append(read_image(forg_paths[0]))\nif len(mask_paths)>0: sample_imgs.append(read_mask(mask_paths[0]))\nshow_images_grid(sample_imgs, titles=['authentic','forged','mask'], cols=3, figsize=(10,4))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T05:03:55.184398Z","iopub.execute_input":"2025-11-25T05:03:55.185128Z","iopub.status.idle":"2025-11-25T05:03:55.80556Z","shell.execute_reply.started":"2025-11-25T05:03:55.185104Z","shell.execute_reply":"2025-11-25T05:03:55.804814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Quick pair checking\npaired = []\nfor p in forg_paths:\n    fname = os.path.basename(p)\n    maskp = os.path.join(TRAIN_MASKS, fname)\n    if os.path.exists(maskp):\n        paired.append((p, maskp))\nprint('Paired forged-mask:', len(paired))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T05:04:25.882174Z","iopub.execute_input":"2025-11-25T05:04:25.882941Z","iopub.status.idle":"2025-11-25T05:04:29.504191Z","shell.execute_reply.started":"2025-11-25T05:04:25.882915Z","shell.execute_reply":"2025-11-25T05:04:29.50358Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 5) Preprocessing & tf.data pipelines\n# =====================\nIMG_SIZE = (256,256)\nBATCH = 8\nAUTOTUNE = tf.data.AUTOTUNE\n\n# Create DataFrame for classification: all images with label 0/1\nrows = []\nfor p in auth_paths:\n    rows.append({'image': p, 'label': 0})\nfor p in forg_paths:\n    rows.append({'image': p, 'label': 1})\n\ndf = pd.DataFrame(rows)\nprint('Total rows for classification:', len(df))\n\ntrain_df, val_df = train_test_split(df, test_size=0.15, random_state=SEED, stratify=df['label'])\n\n# tf.data generator for classification\ndef preprocess_classification(image_path, label):\n    image = tf.io.read_file(image_path)\n    image = tf.image.decode_png(image, channels=3)\n    image = tf.image.resize(image, IMG_SIZE)\n    image = tf.cast(image, tf.float32) / 255.0\n    return image, label\n\n\ndef make_class_dataset(df, batch=BATCH, shuffle=True):\n    ds = tf.data.Dataset.from_tensor_slices((df['image'].values.astype(str), df['label'].values.astype(np.int32)))\n    if shuffle: ds = ds.shuffle(2048, seed=SEED)\n    ds = ds.map(lambda x,y: preprocess_classification(x,y), num_parallel_calls=AUTOTUNE)\n    ds = ds.batch(batch).prefetch(AUTOTUNE)\n    return ds\n\ntrain_ds = make_class_dataset(train_df)\nval_ds = make_class_dataset(val_df, shuffle=False)\n\n# tf.data generator for segmentation (use only forged images with masks)\nseg_pairs = [(p, os.path.join(TRAIN_MASKS, os.path.basename(p))) for p in forg_paths if os.path.exists(os.path.join(TRAIN_MASKS, os.path.basename(p)))]\nseg_df = pd.DataFrame(seg_pairs, columns=['image','mask'])\ntrain_s, val_s = train_test_split(seg_df, test_size=0.15, random_state=SEED)\n\ndef preprocess_seg(image_path, mask_path):\n    image = tf.io.read_file(image_path)\n    image = tf.image.decode_png(image, channels=3)\n    image = tf.image.resize(image, IMG_SIZE)\n    image = tf.cast(image, tf.float32) / 255.0\n\n    mask = tf.io.read_file(mask_path)\n    mask = tf.image.decode_png(mask, channels=1)\n    mask = tf.image.resize(mask, IMG_SIZE, method='nearest')\n    mask = tf.cast(mask > 127, tf.float32)\n    return image, mask\n\n\ndef make_seg_dataset(df, batch=BATCH, shuffle=True):\n    ds = tf.data.Dataset.from_tensor_slices((df['image'].values.astype(str), df['mask'].values.astype(str)))\n    if shuffle: ds = ds.shuffle(1024, seed=SEED)\n    ds = ds.map(lambda x,y: preprocess_seg(x,y), num_parallel_calls=AUTOTUNE)\n    ds = ds.batch(batch).prefetch(AUTOTUNE)\n    return ds\n\ntrain_seg_ds = make_seg_dataset(train_s)\nval_seg_ds = make_seg_dataset(val_s, shuffle=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T05:04:52.867409Z","iopub.execute_input":"2025-11-25T05:04:52.867728Z","iopub.status.idle":"2025-11-25T05:04:59.987251Z","shell.execute_reply.started":"2025-11-25T05:04:52.867708Z","shell.execute_reply":"2025-11-25T05:04:59.986482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install keras --upgrade\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T05:08:24.514099Z","iopub.execute_input":"2025-11-25T05:08:24.514645Z","iopub.status.idle":"2025-11-25T05:08:29.179272Z","shell.execute_reply.started":"2025-11-25T05:08:24.514624Z","shell.execute_reply":"2025-11-25T05:08:29.178564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.applications.efficientnet import EfficientNetB0\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T05:08:36.93928Z","iopub.execute_input":"2025-11-25T05:08:36.93971Z","iopub.status.idle":"2025-11-25T05:08:36.955695Z","shell.execute_reply.started":"2025-11-25T05:08:36.939672Z","shell.execute_reply":"2025-11-25T05:08:36.954981Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nprint(tf.__version__)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T05:08:39.830297Z","iopub.execute_input":"2025-11-25T05:08:39.830907Z","iopub.status.idle":"2025-11-25T05:08:39.834731Z","shell.execute_reply.started":"2025-11-25T05:08:39.830884Z","shell.execute_reply":"2025-11-25T05:08:39.833994Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Standard TF 2.x\nfrom tensorflow.keras.applications import EfficientNetB0\n\n# Alternative if above fails\nfrom tensorflow.keras.applications.efficientnet import EfficientNetB0\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T05:08:44.808314Z","iopub.execute_input":"2025-11-25T05:08:44.809286Z","iopub.status.idle":"2025-11-25T05:08:44.812426Z","shell.execute_reply.started":"2025-11-25T05:08:44.80926Z","shell.execute_reply":"2025-11-25T05:08:44.811866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# =====================\n# 6) Models\n# =====================\n# 6a) Simple CNN classifier (transfer learning using EfficientNetB0)\nfrom tensorflow.keras.applications import EfficientNetB0\n\ndef build_classifier(input_shape=(*IMG_SIZE,3)):\n    base = EfficientNetB0(include_top=False, input_shape=input_shape, weights='imagenet')\n    base.trainable = False\n    inp = layers.Input(shape=input_shape)\n    x = base(inp, training=False)\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Dropout(0.3)(x)\n    x = layers.Dense(128, activation='relu')(x)\n    x = layers.Dropout(0.2)(x)\n    out = layers.Dense(1, activation='sigmoid')(x)\n    model = keras.Model(inp, out)\n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy', keras.metrics.AUC(name='auc')])\n    return model\n\nclf = build_classifier()\nclf.summary()\n\n# 6b) U-Net for segmentation (lightweight)\ndef conv_block(x, filters):\n    x = layers.Conv2D(filters, 3, padding='same', activation='relu')(x)\n    x = layers.Conv2D(filters, 3, padding='same', activation='relu')(x)\n    return x\n\n\ndef build_unet(input_shape=(*IMG_SIZE,3)):\n    inputs = layers.Input(shape=input_shape)\n    # Encoder\n    c1 = conv_block(inputs, 32)\n    p1 = layers.MaxPool2D()(c1)\n\n    c2 = conv_block(p1, 64)\n    p2 = layers.MaxPool2D()(c2)\n\n    c3 = conv_block(p2, 128)\n    p3 = layers.MaxPool2D()(c3)\n\n    c4 = conv_block(p3, 256)\n    p4 = layers.MaxPool2D()(c4)\n\n    # Bridge\n    b = conv_block(p4, 512)\n\n    # Decoder\n    u1 = layers.UpSampling2D()(b)\n    u1 = layers.Concatenate()([u1, c4])\n    c5 = conv_block(u1, 256)\n\n    u2 = layers.UpSampling2D()(c5)\n    u2 = layers.Concatenate()([u2, c3])\n    c6 = conv_block(u2, 128)\n\n    u3 = layers.UpSampling2D()(c6)\n    u3 = layers.Concatenate()([u3, c2])\n    c7 = conv_block(u3, 64)\n\n    u4 = layers.UpSampling2D()(c7)\n    u4 = layers.Concatenate()([u4, c1])\n    c8 = conv_block(u4, 32)\n\n    outputs = layers.Conv2D(1, 1, activation='sigmoid')(c8)\n\n    model = keras.Model(inputs, outputs)\n    model.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy', keras.metrics.MeanIoU(num_classes=2)])\n    return model\n\nunet = build_unet()\nunet.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T05:08:51.465126Z","iopub.execute_input":"2025-11-25T05:08:51.465851Z","iopub.status.idle":"2025-11-25T05:08:54.102545Z","shell.execute_reply.started":"2025-11-25T05:08:51.465827Z","shell.execute_reply":"2025-11-25T05:08:54.101975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow import keras\n\nclf_callbacks = [\n    keras.callbacks.ModelCheckpoint('clf_best.h5', save_best_only=True, monitor='val_auc', mode='max'),\n    keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=3)\n]\n\nhistory_clf = clf.fit(train_ds, validation_data=val_ds, epochs=8, callbacks=clf_callbacks)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T05:18:30.666679Z","iopub.execute_input":"2025-11-25T05:18:30.667437Z","iopub.status.idle":"2025-11-25T05:24:42.663779Z","shell.execute_reply.started":"2025-11-25T05:18:30.667412Z","shell.execute_reply":"2025-11-25T05:24:42.663206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"after edit\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-25T05:28:54.494175Z","iopub.execute_input":"2025-11-25T05:28:54.494959Z","iopub.status.idle":"2025-11-25T05:28:54.498967Z","shell.execute_reply.started":"2025-11-25T05:28:54.494935Z","shell.execute_reply":"2025-11-25T05:28:54.498254Z"}},"outputs":[],"execution_count":null}]}