{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":924245,"sourceType":"datasetVersion","datasetId":464091}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# DeepFake Image detection","metadata":{"id":"vGkKRV1kY8Z-"}},{"cell_type":"code","source":"import sys\nimport sklearn\nimport tensorflow as tf\n\nimport cv2\nimport pandas as pd\nimport numpy as np\n\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nfrom matplotlib import pyplot as plt","metadata":{"id":"TFSU3FCOpKzu","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:54:55.069393Z","iopub.execute_input":"2025-05-25T12:54:55.070075Z","iopub.status.idle":"2025-05-25T12:54:55.079091Z","shell.execute_reply.started":"2025-05-25T12:54:55.070051Z","shell.execute_reply":"2025-05-25T12:54:55.078557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.test.is_gpu_available()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:54:57.869443Z","iopub.execute_input":"2025-05-25T12:54:57.870224Z","iopub.status.idle":"2025-05-25T12:54:58.201855Z","shell.execute_reply.started":"2025-05-25T12:54:57.870189Z","shell.execute_reply":"2025-05-25T12:54:58.201013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.__version__","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.rc('font', size=14)\nplt.rc('axes', labelsize=14, titlesize=14)\nplt.rc('legend', fontsize=14)\nplt.rc('xtick', labelsize=10)\nplt.rc('ytick', labelsize=10)","metadata":{"id":"8d4TH3NbpKzx","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Visualisation","metadata":{"id":"NL3Ht4wC9b3n"}},{"cell_type":"code","source":"import os\n\ndef get_data():\n    return pd.read_csv('../input/deepfake-faces/metadata.csv')","metadata":{"id":"jfv9PxSB4tM8","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:57:40.994506Z","iopub.execute_input":"2025-05-25T12:57:40.995211Z","iopub.status.idle":"2025-05-25T12:57:40.999155Z","shell.execute_reply.started":"2025-05-25T12:57:40.995186Z","shell.execute_reply":"2025-05-25T12:57:40.998468Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta=get_data()\nmeta.head()","metadata":{"id":"tDW7BRph9ehF","outputId":"97de18b5-0a37-4302-8804-8a16a7d2ed2f","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:57:41.295890Z","iopub.execute_input":"2025-05-25T12:57:41.296126Z","iopub.status.idle":"2025-05-25T12:57:41.413102Z","shell.execute_reply.started":"2025-05-25T12:57:41.296107Z","shell.execute_reply":"2025-05-25T12:57:41.412179Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta.shape","metadata":{"id":"n7FSdDifbZxn","outputId":"5451a127-405a-4c0b-a197-c920b796adbb","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:57:42.853979Z","iopub.execute_input":"2025-05-25T12:57:42.854245Z","iopub.status.idle":"2025-05-25T12:57:42.859373Z","shell.execute_reply.started":"2025-05-25T12:57:42.854226Z","shell.execute_reply":"2025-05-25T12:57:42.858773Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(meta[meta.label=='FAKE']),len(meta[meta.label=='REAL'])","metadata":{"id":"_FJcz2IthxVG","outputId":"274c3f65-7acb-4f99-8aa9-a5b2a23bf06a","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:57:44.197123Z","iopub.execute_input":"2025-05-25T12:57:44.197395Z","iopub.status.idle":"2025-05-25T12:57:44.225042Z","shell.execute_reply.started":"2025-05-25T12:57:44.197377Z","shell.execute_reply":"2025-05-25T12:57:44.224155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"real_df = meta[meta[\"label\"] == \"REAL\"]\nfake_df = meta[meta[\"label\"] == \"FAKE\"]\nsample_size = 8000\n\nreal_df = real_df.sample(sample_size, random_state=42)\nfake_df = fake_df.sample(sample_size, random_state=42)\n\nsample_meta = pd.concat([real_df, fake_df])","metadata":{"id":"IgMfzY-PjjtH","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:57:45.433151Z","iopub.execute_input":"2025-05-25T12:57:45.433827Z","iopub.status.idle":"2025-05-25T12:57:45.467318Z","shell.execute_reply.started":"2025-05-25T12:57:45.433802Z","shell.execute_reply":"2025-05-25T12:57:45.466732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nTrain_set, Test_set = train_test_split(sample_meta,test_size=0.2,random_state=42,stratify=sample_meta['label'])\nTrain_set, Val_set  = train_test_split(Train_set,test_size=0.3,random_state=42,stratify=Train_set['label'])","metadata":{"id":"5eB86S6K-T5Z","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:57:47.234024Z","iopub.execute_input":"2025-05-25T12:57:47.234660Z","iopub.status.idle":"2025-05-25T12:57:47.265256Z","shell.execute_reply.started":"2025-05-25T12:57:47.234638Z","shell.execute_reply":"2025-05-25T12:57:47.264658Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Train_set.shape,Val_set.shape,Test_set.shape","metadata":{"id":"8p-TONijb4qA","outputId":"56d0b529-9d81-4019-d8fa-618c8cdba90f","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:57:48.914989Z","iopub.execute_input":"2025-05-25T12:57:48.915796Z","iopub.status.idle":"2025-05-25T12:57:48.920948Z","shell.execute_reply.started":"2025-05-25T12:57:48.915771Z","shell.execute_reply":"2025-05-25T12:57:48.920083Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The original image dataset were biased with more fake images than real since we are taking a sample of it its better to take equal proportion of real and fake images.","metadata":{}},{"cell_type":"code","source":"y = dict()\n\ny[0] = []\ny[1] = []\n\nfor set_name in (np.array(Train_set['label']), np.array(Val_set['label']), np.array(Test_set['label'])):\n    y[0].append(np.sum(set_name == 'REAL'))\n    y[1].append(np.sum(set_name == 'FAKE'))\n\ntrace0 = go.Bar(\n    x=['Train Set', 'Validation Set', 'Test Set'],\n    y=y[0],\n    name='REAL',\n    marker=dict(color='#33cc33'),\n    opacity=0.7\n)\ntrace1 = go.Bar(\n    x=['Train Set', 'Validation Set', 'Test Set'],\n    y=y[1],\n    name='FAKE',\n    marker=dict(color='#ff3300'),\n    opacity=0.7\n)\n\ndata = [trace0, trace1]\nlayout = go.Layout(\n    title='Count of classes in each set',\n    xaxis={'title': 'Set'},\n    yaxis={'title': 'Count'}\n)\n\nfig = go.Figure(data, layout)\niplot(fig)","metadata":{"id":"hzNGtCWd-mTk","outputId":"5178c3ed-cbba-4f99-99d0-26bde11a5dab","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:57:50.995469Z","iopub.execute_input":"2025-05-25T12:57:50.995818Z","iopub.status.idle":"2025-05-25T12:57:51.564190Z","shell.execute_reply.started":"2025-05-25T12:57:50.995796Z","shell.execute_reply":"2025-05-25T12:57:51.563496Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(15,15))\nfor cur,i in enumerate(Train_set.index[25:50]):\n    plt.subplot(5,5,cur+1)\n    plt.xticks([])\n    plt.yticks([])\n    plt.grid(False)\n    \n    plt.imshow(cv2.imread('../input/deepfake-faces/faces_224/'+Train_set.loc[i,'videoname'][:-4]+'.jpg'))\n    \n    if(Train_set.loc[i,'label']=='FAKE'):\n        plt.xlabel('FAKE Image')\n    else:\n        plt.xlabel('REAL Image')\n        \nplt.show()","metadata":{"id":"VR7Uly2fcUYi","outputId":"c1f47a82-ef4f-4bcd-b51c-d4738142fc0f","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:57:52.273490Z","iopub.execute_input":"2025-05-25T12:57:52.273733Z","iopub.status.idle":"2025-05-25T12:57:53.798370Z","shell.execute_reply.started":"2025-05-25T12:57:52.273718Z","shell.execute_reply":"2025-05-25T12:57:53.797482Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Modelling","metadata":{"id":"dOvN_divkl-N"}},{"cell_type":"markdown","source":"Before jumping to use pretrained model lets develop some base line model to test how our pretrained model outperforms.","metadata":{}},{"cell_type":"markdown","source":"### Custom CNN Architecture","metadata":{"id":"oid44Xx-pKz6"}},{"cell_type":"code","source":"def retreive_dataset(set_name):\n    images,labels=[],[]\n    for (img, imclass) in zip(set_name['videoname'], set_name['label']):\n        images.append(cv2.imread('../input/deepfake-faces/faces_224/'+img[:-4]+'.jpg'))\n        if(imclass=='FAKE'):\n            labels.append(1)\n        else:\n            labels.append(0)\n    \n    return np.array(images),np.array(labels)","metadata":{"id":"Hz0ZdQ_fgHhG","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:57:57.756197Z","iopub.execute_input":"2025-05-25T12:57:57.756952Z","iopub.status.idle":"2025-05-25T12:57:57.761695Z","shell.execute_reply.started":"2025-05-25T12:57:57.756927Z","shell.execute_reply":"2025-05-25T12:57:57.760829Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train,y_train=retreive_dataset(Train_set)\nX_val,y_val=retreive_dataset(Val_set)\nX_test,y_test=retreive_dataset(Test_set)","metadata":{"id":"zeAGRcAbguKU","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:57:58.201287Z","iopub.execute_input":"2025-05-25T12:57:58.201600Z","iopub.status.idle":"2025-05-25T12:58:42.391535Z","shell.execute_reply.started":"2025-05-25T12:57:58.201573Z","shell.execute_reply":"2025-05-25T12:58:42.390870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from functools import partial\n\ntf.random.set_seed(42) \nDefaultConv2D = partial(tf.keras.layers.Conv2D, kernel_size=3, padding=\"same\",\n                        activation=\"relu\", kernel_initializer=\"he_normal\")\n\nmodel = tf.keras.Sequential([\n    DefaultConv2D(filters=64, kernel_size=7, input_shape=[224, 224, 3]),\n    tf.keras.layers.MaxPool2D(),\n    DefaultConv2D(filters=128),\n    DefaultConv2D(filters=128),\n    tf.keras.layers.MaxPool2D(),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(units=128, activation=\"relu\",\n                          kernel_initializer=\"he_normal\"),\n    tf.keras.layers.Dropout(0.5),\n    tf.keras.layers.Dense(units=64, activation=\"relu\",\n                          kernel_initializer=\"he_normal\"),\n    tf.keras.layers.Dropout(0.5),\n    tf.keras.layers.Dense(units=1, activation=\"sigmoid\")\n])","metadata":{"id":"34upiak4pKz6","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:58:42.392937Z","iopub.execute_input":"2025-05-25T12:58:42.393650Z","iopub.status.idle":"2025-05-25T12:58:43.248158Z","shell.execute_reply.started":"2025-05-25T12:58:42.393627Z","shell.execute_reply":"2025-05-25T12:58:43.247374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(loss=\"binary_crossentropy\", optimizer=\"nadam\",\n              metrics=[\"accuracy\"])\nmodel.summary()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs=10,batch_size=64,\n                    validation_data=(X_val, y_val))","metadata":{"id":"KZbWeIBYpKz6","outputId":"deb6f56a-7b93-4241-a1bd-b210c0f2d426","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score = model.evaluate(X_test, y_test)","metadata":{"id":"6HDDr4uehast","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot model performance\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs_range = range(1, len(history.epoch) + 1)\n\nplt.figure(figsize=(15,5))\n\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Train Set')\nplt.plot(epochs_range, val_acc, label='Val Set')\nplt.legend(loc=\"best\")\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.title('Model Accuracy')\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Train Set')\nplt.plot(epochs_range, val_loss, label='Val Set')\nplt.legend(loc=\"best\")\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.title('Model Loss')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Pretrained Models for Transfer Learning","metadata":{"id":"hqxnSBJ3pKz8"}},{"cell_type":"markdown","source":"Here we used Xception model for fine-tuning feel free to try the performance of other pretrained models.","metadata":{}},{"cell_type":"markdown","source":"All three datasets contain individual images. We need to batch them, but for this we first need to ensure they all have the same size, or else batching will not work. We can use a `Resizing` layer for this. We must also call the `tf.keras.applications.xception.preprocess_input()` function to preprocess the images appropriately for the Xception model. We will also add shuffling and prefetching to the training dataset.","metadata":{"id":"gXG6iv8XpKz9"}},{"cell_type":"code","source":"train_set_raw=tf.data.Dataset.from_tensor_slices((X_train,y_train))\nvalid_set_raw=tf.data.Dataset.from_tensor_slices((X_val,y_val))\ntest_set_raw=tf.data.Dataset.from_tensor_slices((X_test,y_test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:58:43.249025Z","iopub.execute_input":"2025-05-25T12:58:43.249266Z","iopub.status.idle":"2025-05-25T12:58:49.339771Z","shell.execute_reply.started":"2025-05-25T12:58:43.249250Z","shell.execute_reply":"2025-05-25T12:58:49.339181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.backend.clear_session()  # extra code – resets layer name counter\n\nbatch_size = 32\npreprocess = tf.keras.applications.xception.preprocess_input\ntrain_set = train_set_raw.map(lambda X, y: (preprocess(tf.cast(X, tf.float32)), y))\ntrain_set = train_set.shuffle(1000, seed=42).batch(batch_size).prefetch(1)\nvalid_set = valid_set_raw.map(lambda X, y: (preprocess(tf.cast(X, tf.float32)), y)).batch(batch_size)\ntest_set = test_set_raw.map(lambda X, y: (preprocess(tf.cast(X, tf.float32)), y)).batch(batch_size)","metadata":{"id":"Bnz0n9XApKz9","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:58:49.341248Z","iopub.execute_input":"2025-05-25T12:58:49.341495Z","iopub.status.idle":"2025-05-25T12:58:50.847665Z","shell.execute_reply.started":"2025-05-25T12:58:49.341478Z","shell.execute_reply":"2025-05-25T12:58:50.847090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extra code – displays the first 9 images in the first batch of valid_set\n\nplt.figure(figsize=(12, 12))\nfor X_batch, y_batch in valid_set.take(1):\n    for index in range(9):\n        plt.subplot(3, 3, index + 1)\n        plt.imshow((X_batch[index] + 1) / 2)  # rescale to 0–1 for imshow()\n        if(y_batch[index]==1):\n            classt='FAKE'\n        else:\n            classt='REAL'\n        plt.title(f\"Class: {classt}\")\n        plt.axis(\"off\")\n\nplt.show()","metadata":{"id":"ZL3c3i4opKz9","outputId":"38847d8d-8822-41a3-cfb2-27479aa5debe","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:58:50.848308Z","iopub.execute_input":"2025-05-25T12:58:50.848480Z","iopub.status.idle":"2025-05-25T12:58:52.327357Z","shell.execute_reply.started":"2025-05-25T12:58:50.848467Z","shell.execute_reply":"2025-05-25T12:58:52.326598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_augmentation = tf.keras.Sequential([\n    tf.keras.layers.RandomFlip(mode=\"horizontal\", seed=42),\n    tf.keras.layers.RandomRotation(factor=0.05, seed=42),\n    tf.keras.layers.RandomContrast(factor=0.2, seed=42)\n])","metadata":{"id":"Ib0cA8Y1pKz9","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:58:52.328147Z","iopub.execute_input":"2025-05-25T12:58:52.328385Z","iopub.status.idle":"2025-05-25T12:58:52.348101Z","shell.execute_reply.started":"2025-05-25T12:58:52.328366Z","shell.execute_reply":"2025-05-25T12:58:52.347335Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extra code – displays the same first 9 images, after augmentation\n\nplt.figure(figsize=(12, 12))\nfor X_batch, y_batch in valid_set.take(1):\n    X_batch_augmented = data_augmentation(X_batch, training=True)\n    for index in range(9):\n        plt.subplot(3, 3, index + 1)\n        # We must rescale the images to the 0-1 range for imshow(), and also\n        # clip the result to that range, because data augmentation may\n        # make some values go out of bounds (e.g., RandomContrast in this case).\n        plt.imshow(np.clip((X_batch_augmented[index] + 1) / 2, 0, 1))\n        if(y_batch[index]==1):\n            classt='FAKE'\n        else:\n            classt='REAL'\n        plt.title(f\"Class: {classt}\")\n        plt.axis(\"off\")\n\nplt.show()","metadata":{"id":"w6GH5_vupKz-","outputId":"eeb2c924-2f4f-4aa1-bea9-951bebef4bf0","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:58:52.348903Z","iopub.execute_input":"2025-05-25T12:58:52.349287Z","iopub.status.idle":"2025-05-25T12:58:55.238546Z","shell.execute_reply.started":"2025-05-25T12:58:52.349256Z","shell.execute_reply":"2025-05-25T12:58:55.237724Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now let's load the pretrained model, without its top layers, and replace them with our own task","metadata":{"id":"kNL9AOsDpKz-"}},{"cell_type":"code","source":"tf.random.set_seed(42)  # extra code – ensures reproducibility\nbase_model = tf.keras.applications.xception.Xception(weights=\"imagenet\",\n                                                     include_top=False)\navg = tf.keras.layers.GlobalAveragePooling2D()(base_model.output)\noutput = tf.keras.layers.Dense(1, activation=\"sigmoid\")(avg)\nmodel = tf.keras.Model(inputs=base_model.input, outputs=output)","metadata":{"id":"lRyCgvaKpKz-","outputId":"a825e173-8b1d-4217-a1c4-5491b49c3e82","trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:58:55.239435Z","iopub.execute_input":"2025-05-25T12:58:55.239808Z","iopub.status.idle":"2025-05-25T12:58:56.383164Z","shell.execute_reply.started":"2025-05-25T12:58:55.239781Z","shell.execute_reply":"2025-05-25T12:58:56.382301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for layer in base_model.layers:\n    layer.trainable = False","metadata":{"id":"KBlyG6ElpKz-","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer = tf.keras.optimizers.SGD(learning_rate=0.1, momentum=0.9)\nmodel.compile(loss=\"binary_crossentropy\", optimizer=optimizer,\n              metrics=[\"accuracy\"])\nhistory = model.fit(train_set, validation_data=valid_set, epochs=3)","metadata":{"id":"GGxK2yPcpKz-","outputId":"6b64214a-e104-4b6c-9b7a-3388fc9aa15f","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for indices in zip(range(33), range(33, 66), range(66, 99), range(99, 132)):\n    for idx in indices:\n        print(f\"{idx:3}: {base_model.layers[idx].name:22}\", end=\"\")\n    print()","metadata":{"id":"GvGMiJMLpKz-","outputId":"91f2c96c-c058-45e0-e428-66fa6076ad56","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(test_set)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping\n\nfor layer in base_model.layers[56:]:\n    layer.trainable = True\n\n\n# Define the EarlyStopping callback\nearly_stopping = EarlyStopping(monitor='val_loss', patience=5, restore_best_weights=True)\n\noptimizer = tf.keras.optimizers.SGD(learning_rate=0.01, momentum=0.9)\nmodel.compile(loss=\"binary_crossentropy\", optimizer=optimizer,\n              metrics=[\"accuracy\"])\nhistory = model.fit(train_set, validation_data=valid_set, epochs=20, callbacks=[early_stopping])","metadata":{"id":"GEUNGlhvpKz_","outputId":"c622a91d-f634-4443-b87e-8d46defdb578","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot model performance\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs_range = range(1, len(history.epoch) + 1)\n\nplt.figure(figsize=(15,5))\n\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Train Set')\nplt.plot(epochs_range, val_acc, label='Val Set')\nplt.legend(loc=\"best\")\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.title('Model Accuracy')\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Train Set')\nplt.plot(epochs_range, val_loss, label='Val Set')\nplt.legend(loc=\"best\")\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.title('Model Loss')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(test_set)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save('xception_deepfake_image.h5')","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Add Explainability to the model","metadata":{}},{"cell_type":"markdown","source":"Lets try to interpret the trained model on how it finds a image FAKE","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.models import load_model\n\n# Load the saved model\nmodel = load_model('xception_deepfake_image.h5')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:58:56.384377Z","iopub.execute_input":"2025-05-25T12:58:56.385224Z","iopub.status.idle":"2025-05-25T12:58:57.193077Z","shell.execute_reply.started":"2025-05-25T12:58:56.385195Z","shell.execute_reply":"2025-05-25T12:58:57.192452Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install lime","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:59:09.247857Z","iopub.execute_input":"2025-05-25T12:59:09.248415Z","iopub.status.idle":"2025-05-25T12:59:12.657165Z","shell.execute_reply.started":"2025-05-25T12:59:09.248395Z","shell.execute_reply":"2025-05-25T12:59:12.656161Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lime import lime_image\nimport matplotlib.pyplot as plt\n\nexplainer = lime_image.LimeImageExplainer()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:59:12.658882Z","iopub.execute_input":"2025-05-25T12:59:12.659163Z","iopub.status.idle":"2025-05-25T12:59:12.705218Z","shell.execute_reply.started":"2025-05-25T12:59:12.659138Z","shell.execute_reply":"2025-05-25T12:59:12.704373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 12))\nfor X_batch, y_batch in test_set.take(1):\n    X_batch_augmented = data_augmentation(X_batch, training=True)\n    for index in range(9):\n        plt.subplot(3, 3, index + 1)\n        plt.imshow(np.clip((X_batch_augmented[index] + 1) / 2, 0, 1))\n        if(y_batch[index]==1):\n            classt='FAKE'\n        else:\n            classt='REAL'\n        plt.title(f\"Class: {classt}\")\n        plt.axis(\"off\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:59:12.706034Z","iopub.execute_input":"2025-05-25T12:59:12.707196Z","iopub.status.idle":"2025-05-25T12:59:14.029530Z","shell.execute_reply.started":"2025-05-25T12:59:12.707173Z","shell.execute_reply":"2025-05-25T12:59:14.028816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = X_test[2,:,:,:]\nfor X_batch, y_batch in test_set.take(1):\n    test_data = data_augmentation(X_batch, training=True)\n\ntest_data = np.array(test_data[2,:,:,:])\ntest_data.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:59:14.030938Z","iopub.execute_input":"2025-05-25T12:59:14.031440Z","iopub.status.idle":"2025-05-25T12:59:14.506367Z","shell.execute_reply.started":"2025-05-25T12:59:14.031400Z","shell.execute_reply":"2025-05-25T12:59:14.505534Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"explanation = explainer.explain_instance(test_data.astype('double'), model.predict,  \n                                         top_labels=3, hide_color=0, num_samples=1000)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:59:14.507249Z","iopub.execute_input":"2025-05-25T12:59:14.507557Z","iopub.status.idle":"2025-05-25T12:59:39.610600Z","shell.execute_reply.started":"2025-05-25T12:59:14.507530Z","shell.execute_reply":"2025-05-25T12:59:39.609991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from skimage.segmentation import mark_boundaries\n\ntemp_1, mask_1 = explanation.get_image_and_mask(explanation.top_labels[0], positive_only=True, num_features=5, hide_rest=True)\ntemp_2, mask_2 = explanation.get_image_and_mask(explanation.top_labels[0], positive_only=False, num_features=10, hide_rest=False)\n\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(15,15))\nax1.imshow(mark_boundaries(temp_1, mask_1))\nax2.imshow(mark_boundaries(temp_2, mask_2))\nax1.axis('off')\nax2.axis('off')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:59:39.611959Z","iopub.execute_input":"2025-05-25T12:59:39.612420Z","iopub.status.idle":"2025-05-25T12:59:39.911351Z","shell.execute_reply.started":"2025-05-25T12:59:39.612401Z","shell.execute_reply":"2025-05-25T12:59:39.910641Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shap\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\n\n# Load and prepare your data\nfor X_batch, y_batch in test_set.take(1):\n    X_images = data_augmentation(X_batch, training=True).numpy()\n    X_sample = X_images[2:3]  # 3rd image (keep as batch of 1 for model compatibility)\n    y_true = y_batch[2].numpy()  # True label for reference\n\n# Verify shapes\nprint(\"Image shape:\", X_sample.shape)  # Should be (1, H, W, C)\nprint(\"True label:\", \"REAL\" if y_true == 0 else \"FAKE\")\n\n# Create masker and explainer\nmasker = shap.maskers.Image(\"inpaint_telea\", X_sample[0].shape)\nexplainer = shap.Explainer(model.predict, masker, output_names=[\"REAL\", \"FAKE\"])\n\n# Compute SHAP values with conservative parameters\nshap_values = explainer(\n    X_sample,\n    max_evals=500,  # Reduced from default 1000\n    batch_size=5,    # Small batches to avoid OOM\n    silent=True      # Suppress verbose output\n)\n\n# Visualization - consistent plotting\nplt.figure(figsize=(12, 6))\n\n# Initialize pred_class before the if block\npred_class = None\n\n# For binary classification\nif isinstance(shap_values, list) and len(shap_values) == 2:\n    # Get predicted class\n    pred_probs = model.predict(X_sample)\n    pred_class = np.argmax(pred_probs[0])\n    \n    # Plot only the explanation for the predicted class\n    shap.image_plot(\n        [shap_values[pred_class][0]],  # Note double list [[...]] for correct dimensions\n        X_sample,\n        labels=[[[\"REAL\", \"FAKE\"][pred_class]]],\n        show=False\n    )\nelse:\n    # Fallback for non-binary cases\n    shap.image_plot(shap_values, X_sample, show=False)\n    # For non-binary case, we'll just show true label\n    pred_class = y_true  # Assuming y_true matches your model's class indices\n\n# Fix: Check if pred_class was defined before using it\ntitle = f\"SHAP Explanation\\nTrue: {'REAL' if y_true == 0 else 'FAKE'}\"\nif pred_class is not None:\n    title += f\" | Predicted: {['REAL', 'FAKE'][pred_class]}\"\n\nplt.title(title, pad=20)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-25T12:59:39.912330Z","iopub.execute_input":"2025-05-25T12:59:39.912910Z","iopub.status.idle":"2025-05-25T13:00:27.407409Z","shell.execute_reply.started":"2025-05-25T12:59:39.912884Z","shell.execute_reply":"2025-05-25T13:00:27.406639Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}