{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":924245,"sourceType":"datasetVersion","datasetId":464091}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# DeepFake Image detection","metadata":{"id":"vGkKRV1kY8Z-"}},{"cell_type":"code","source":"import sys\nimport sklearn\nimport tensorflow as tf\n\nimport cv2\nimport pandas as pd\nimport numpy as np\n\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nfrom matplotlib import pyplot as plt","metadata":{"id":"TFSU3FCOpKzu","execution":{"iopub.status.busy":"2025-05-27T18:23:47.212800Z","iopub.execute_input":"2025-05-27T18:23:47.213282Z","iopub.status.idle":"2025-05-27T18:24:01.701106Z","shell.execute_reply.started":"2025-05-27T18:23:47.213257Z","shell.execute_reply":"2025-05-27T18:24:01.700583Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.test.is_gpu_available()","metadata":{"execution":{"iopub.status.busy":"2025-05-27T18:24:01.749176Z","iopub.execute_input":"2025-05-27T18:24:01.749378Z","iopub.status.idle":"2025-05-27T18:24:02.973714Z","shell.execute_reply.started":"2025-05-27T18:24:01.749361Z","shell.execute_reply":"2025-05-27T18:24:02.972894Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.__version__","metadata":{"execution":{"iopub.status.busy":"2025-05-27T18:24:15.368271Z","iopub.execute_input":"2025-05-27T18:24:15.368821Z","iopub.status.idle":"2025-05-27T18:24:15.373950Z","shell.execute_reply.started":"2025-05-27T18:24:15.368796Z","shell.execute_reply":"2025-05-27T18:24:15.373167Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.rc('font', size=14)\nplt.rc('axes', labelsize=14, titlesize=14)\nplt.rc('legend', fontsize=14)\nplt.rc('xtick', labelsize=10)\nplt.rc('ytick', labelsize=10)","metadata":{"id":"8d4TH3NbpKzx","execution":{"iopub.status.busy":"2025-05-27T18:48:00.415242Z","iopub.execute_input":"2025-05-27T18:48:00.415558Z","iopub.status.idle":"2025-05-27T18:48:00.419928Z","shell.execute_reply.started":"2025-05-27T18:48:00.415527Z","shell.execute_reply":"2025-05-27T18:48:00.419235Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Visualisation","metadata":{"id":"NL3Ht4wC9b3n"}},{"cell_type":"code","source":"import os\n\ndef get_data():\n    return pd.read_csv('../input/deepfake-faces/metadata.csv')","metadata":{"id":"jfv9PxSB4tM8","execution":{"iopub.status.busy":"2025-05-27T18:46:53.968779Z","iopub.execute_input":"2025-05-27T18:46:53.969065Z","iopub.status.idle":"2025-05-27T18:46:53.972979Z","shell.execute_reply.started":"2025-05-27T18:46:53.969044Z","shell.execute_reply":"2025-05-27T18:46:53.972189Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta=get_data()\nmeta.head()","metadata":{"id":"tDW7BRph9ehF","outputId":"97de18b5-0a37-4302-8804-8a16a7d2ed2f","execution":{"iopub.status.busy":"2025-05-27T18:24:38.273109Z","iopub.execute_input":"2025-05-27T18:24:38.273791Z","iopub.status.idle":"2025-05-27T18:24:38.497966Z","shell.execute_reply.started":"2025-05-27T18:24:38.273756Z","shell.execute_reply":"2025-05-27T18:24:38.497094Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta.shape","metadata":{"id":"n7FSdDifbZxn","outputId":"5451a127-405a-4c0b-a197-c920b796adbb","execution":{"iopub.status.busy":"2025-05-27T18:24:47.435074Z","iopub.execute_input":"2025-05-27T18:24:47.435678Z","iopub.status.idle":"2025-05-27T18:24:47.440665Z","shell.execute_reply.started":"2025-05-27T18:24:47.435654Z","shell.execute_reply":"2025-05-27T18:24:47.439749Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(meta[meta.label=='FAKE']),len(meta[meta.label=='REAL'])","metadata":{"id":"_FJcz2IthxVG","outputId":"274c3f65-7acb-4f99-8aa9-a5b2a23bf06a","execution":{"iopub.status.busy":"2025-05-27T18:24:50.339256Z","iopub.execute_input":"2025-05-27T18:24:50.339779Z","iopub.status.idle":"2025-05-27T18:24:50.370741Z","shell.execute_reply.started":"2025-05-27T18:24:50.339753Z","shell.execute_reply":"2025-05-27T18:24:50.370143Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"real_df = meta[meta[\"label\"] == \"REAL\"]\nfake_df = meta[meta[\"label\"] == \"FAKE\"]\nsample_size = 8000\n\nreal_df = real_df.sample(sample_size, random_state=42)\nfake_df = fake_df.sample(sample_size, random_state=42)\n\nsample_meta = pd.concat([real_df, fake_df])","metadata":{"id":"IgMfzY-PjjtH","execution":{"iopub.status.busy":"2025-05-27T18:25:20.648762Z","iopub.execute_input":"2025-05-27T18:25:20.649033Z","iopub.status.idle":"2025-05-27T18:25:20.679335Z","shell.execute_reply.started":"2025-05-27T18:25:20.649012Z","shell.execute_reply":"2025-05-27T18:25:20.678806Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nTrain_set, Test_set = train_test_split(sample_meta,test_size=0.2,random_state=42,stratify=sample_meta['label'])\nTrain_set, Val_set  = train_test_split(Train_set,test_size=0.3,random_state=42,stratify=Train_set['label'])","metadata":{"id":"5eB86S6K-T5Z","execution":{"iopub.status.busy":"2025-05-27T18:25:22.800574Z","iopub.execute_input":"2025-05-27T18:25:22.801194Z","iopub.status.idle":"2025-05-27T18:25:22.935091Z","shell.execute_reply.started":"2025-05-27T18:25:22.801172Z","shell.execute_reply":"2025-05-27T18:25:22.934581Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Train_set.shape,Val_set.shape,Test_set.shape","metadata":{"id":"8p-TONijb4qA","outputId":"56d0b529-9d81-4019-d8fa-618c8cdba90f","execution":{"iopub.status.busy":"2025-05-27T18:25:29.031209Z","iopub.execute_input":"2025-05-27T18:25:29.031805Z","iopub.status.idle":"2025-05-27T18:25:29.036917Z","shell.execute_reply.started":"2025-05-27T18:25:29.031779Z","shell.execute_reply":"2025-05-27T18:25:29.036168Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = dict()\n\ny[0] = []\ny[1] = []\n\nfor set_name in (np.array(Train_set['label']), np.array(Val_set['label']), np.array(Test_set['label'])):\n    y[0].append(np.sum(set_name == 'REAL'))\n    y[1].append(np.sum(set_name == 'FAKE'))\n\ntrace0 = go.Bar(\n    x=['Train Set', 'Validation Set', 'Test Set'],\n    y=y[0],\n    name='REAL',\n    marker=dict(color='#33cc33'),\n    opacity=0.7\n)\ntrace1 = go.Bar(\n    x=['Train Set', 'Validation Set', 'Test Set'],\n    y=y[1],\n    name='FAKE',\n    marker=dict(color='#ff3300'),\n    opacity=0.7\n)\n\ndata = [trace0, trace1]\nlayout = go.Layout(\n    title='Count of classes in each set',\n    xaxis={'title': 'Set'},\n    yaxis={'title': 'Count'}\n)\n\nfig = go.Figure(data, layout)\niplot(fig)","metadata":{"id":"hzNGtCWd-mTk","outputId":"5178c3ed-cbba-4f99-99d0-26bde11a5dab","execution":{"iopub.status.busy":"2025-05-27T18:25:55.523759Z","iopub.execute_input":"2025-05-27T18:25:55.524439Z","iopub.status.idle":"2025-05-27T18:25:55.547295Z","shell.execute_reply.started":"2025-05-27T18:25:55.524389Z","shell.execute_reply":"2025-05-27T18:25:55.546763Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The original image dataset were biased with more fake images than real since we are taking a sample of it its better to take equal proportion of real and fake images.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,15))\nfor cur,i in enumerate(Train_set.index[25:50]):\n    plt.subplot(5,5,cur+1)\n    plt.xticks([])\n    plt.yticks([])\n    plt.grid(False)\n    \n    plt.imshow(cv2.imread('../input/deepfake-faces/faces_224/'+Train_set.loc[i,'videoname'][:-4]+'.jpg'))\n    \n    if(Train_set.loc[i,'label']=='FAKE'):\n        plt.xlabel('FAKE Image')\n    else:\n        plt.xlabel('REAL Image')\n        \nplt.show()","metadata":{"id":"VR7Uly2fcUYi","outputId":"c1f47a82-ef4f-4bcd-b51c-d4738142fc0f","execution":{"iopub.status.busy":"2025-05-27T18:26:34.360381Z","iopub.execute_input":"2025-05-27T18:26:34.361085Z","iopub.status.idle":"2025-05-27T18:26:35.743504Z","shell.execute_reply.started":"2025-05-27T18:26:34.361062Z","shell.execute_reply":"2025-05-27T18:26:35.742134Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Modelling","metadata":{"id":"dOvN_divkl-N"}},{"cell_type":"markdown","source":"Before jumping to use pretrained model lets develop some base line model to test how our pretrained model outperforms.","metadata":{}},{"cell_type":"markdown","source":"### Custom CNN Architecture","metadata":{"id":"oid44Xx-pKz6"}},{"cell_type":"code","source":"def retreive_dataset(set_name):\n    images,labels=[],[]\n    for (img, imclass) in zip(set_name['videoname'], set_name['label']):\n        images.append(cv2.imread('../input/deepfake-faces/faces_224/'+img[:-4]+'.jpg'))\n        if(imclass=='FAKE'):\n            labels.append(1)\n        else:\n            labels.append(0)\n    \n    return np.array(images),np.array(labels)","metadata":{"id":"Hz0ZdQ_fgHhG","execution":{"iopub.status.busy":"2025-05-27T18:26:50.377745Z","iopub.execute_input":"2025-05-27T18:26:50.378315Z","iopub.status.idle":"2025-05-27T18:26:50.382887Z","shell.execute_reply.started":"2025-05-27T18:26:50.378284Z","shell.execute_reply":"2025-05-27T18:26:50.382014Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train,y_train=retreive_dataset(Train_set)\nX_val,y_val=retreive_dataset(Val_set)\nX_test,y_test=retreive_dataset(Test_set)","metadata":{"id":"zeAGRcAbguKU","execution":{"iopub.status.busy":"2025-05-27T18:26:55.517538Z","iopub.execute_input":"2025-05-27T18:26:55.518220Z","iopub.status.idle":"2025-05-27T18:29:33.291284Z","shell.execute_reply.started":"2025-05-27T18:26:55.518194Z","shell.execute_reply":"2025-05-27T18:29:33.290606Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from functools import partial\n\ntf.random.set_seed(42) \nDefaultConv2D = partial(tf.keras.layers.Conv2D, kernel_size=3, padding=\"same\",\n                        activation=\"relu\", kernel_initializer=\"he_normal\")\n\nmodel = tf.keras.Sequential([\n    DefaultConv2D(filters=64, kernel_size=7, input_shape=[224, 224, 3]),\n    tf.keras.layers.MaxPool2D(),\n    DefaultConv2D(filters=128),\n    DefaultConv2D(filters=128),\n    tf.keras.layers.MaxPool2D(),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(units=128, activation=\"relu\",\n                          kernel_initializer=\"he_normal\"),\n    tf.keras.layers.Dropout(0.5),\n    tf.keras.layers.Dense(units=64, activation=\"relu\",\n                          kernel_initializer=\"he_normal\"),\n    tf.keras.layers.Dropout(0.5),\n    tf.keras.layers.Dense(units=1, activation=\"sigmoid\")\n])","metadata":{"id":"34upiak4pKz6","execution":{"iopub.status.busy":"2023-04-07T16:28:09.763881Z","iopub.execute_input":"2023-04-07T16:28:09.76461Z","iopub.status.idle":"2023-04-07T16:28:10.157145Z","shell.execute_reply.started":"2023-04-07T16:28:09.764558Z","shell.execute_reply":"2023-04-07T16:28:10.155932Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(loss=\"binary_crossentropy\", optimizer=\"nadam\",\n              metrics=[\"accuracy\"])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T16:28:10.158549Z","iopub.execute_input":"2023-04-07T16:28:10.158965Z","iopub.status.idle":"2023-04-07T16:28:10.213507Z","shell.execute_reply.started":"2023-04-07T16:28:10.158919Z","shell.execute_reply":"2023-04-07T16:28:10.212521Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs=5,batch_size=64,\n                    validation_data=(X_val, y_val))","metadata":{"id":"KZbWeIBYpKz6","outputId":"deb6f56a-7b93-4241-a1bd-b210c0f2d426","execution":{"iopub.status.busy":"2023-04-07T16:28:10.21485Z","iopub.execute_input":"2023-04-07T16:28:10.215286Z","iopub.status.idle":"2023-04-07T16:31:57.928965Z","shell.execute_reply.started":"2023-04-07T16:28:10.215238Z","shell.execute_reply":"2023-04-07T16:31:57.927847Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"score = model.evaluate(X_test, y_test)","metadata":{"id":"6HDDr4uehast","execution":{"iopub.status.busy":"2023-04-07T16:31:57.930699Z","iopub.execute_input":"2023-04-07T16:31:57.931088Z","iopub.status.idle":"2023-04-07T16:32:09.297484Z","shell.execute_reply.started":"2023-04-07T16:31:57.931058Z","shell.execute_reply":"2023-04-07T16:32:09.296254Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot model performance\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs_range = range(1, len(history.epoch) + 1)\n\nplt.figure(figsize=(15,5))\n\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Train Set')\nplt.plot(epochs_range, val_acc, label='Val Set')\nplt.legend(loc=\"best\")\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.title('Model Accuracy')\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Train Set')\nplt.plot(epochs_range, val_loss, label='Val Set')\nplt.legend(loc=\"best\")\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.title('Model Loss')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T16:32:09.299584Z","iopub.execute_input":"2023-04-07T16:32:09.300077Z","iopub.status.idle":"2023-04-07T16:32:09.836573Z","shell.execute_reply.started":"2023-04-07T16:32:09.300028Z","shell.execute_reply":"2023-04-07T16:32:09.835539Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Pretrained Models for Transfer Learning","metadata":{"id":"hqxnSBJ3pKz8"}},{"cell_type":"markdown","source":"Here we used Xception model for fine-tuning feel free to try the performance of other pretrained models.","metadata":{}},{"cell_type":"markdown","source":"All three datasets contain individual images. We need to batch them, but for this we first need to ensure they all have the same size, or else batching will not work. We can use a `Resizing` layer for this. We must also call the `tf.keras.applications.xception.preprocess_input()` function to preprocess the images appropriately for the Xception model. We will also add shuffling and prefetching to the training dataset.","metadata":{"id":"gXG6iv8XpKz9"}},{"cell_type":"code","source":"train_set_raw=tf.data.Dataset.from_tensor_slices((X_train,y_train))\nvalid_set_raw=tf.data.Dataset.from_tensor_slices((X_val,y_val))\ntest_set_raw=tf.data.Dataset.from_tensor_slices((X_test,y_test))","metadata":{"execution":{"iopub.status.busy":"2023-04-07T16:32:09.838195Z","iopub.execute_input":"2023-04-07T16:32:09.838536Z","iopub.status.idle":"2023-04-07T16:32:14.761195Z","shell.execute_reply.started":"2023-04-07T16:32:09.838504Z","shell.execute_reply":"2023-04-07T16:32:14.759712Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.backend.clear_session()  # extra code – resets layer name counter\n\nbatch_size = 32\npreprocess = tf.keras.applications.xception.preprocess_input\ntrain_set = train_set_raw.map(lambda X, y: (preprocess(tf.cast(X, tf.float32)), y))\ntrain_set = train_set.shuffle(1000, seed=42).batch(batch_size).prefetch(1)\nvalid_set = valid_set_raw.map(lambda X, y: (preprocess(tf.cast(X, tf.float32)), y)).batch(batch_size)\ntest_set = test_set_raw.map(lambda X, y: (preprocess(tf.cast(X, tf.float32)), y)).batch(batch_size)","metadata":{"id":"Bnz0n9XApKz9","execution":{"iopub.status.busy":"2023-04-07T16:32:14.764355Z","iopub.execute_input":"2023-04-07T16:32:14.765293Z","iopub.status.idle":"2023-04-07T16:32:14.91687Z","shell.execute_reply.started":"2023-04-07T16:32:14.765241Z","shell.execute_reply":"2023-04-07T16:32:14.915745Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extra code – displays the first 9 images in the first batch of valid_set\n\nplt.figure(figsize=(12, 12))\nfor X_batch, y_batch in valid_set.take(1):\n    for index in range(9):\n        plt.subplot(3, 3, index + 1)\n        plt.imshow((X_batch[index] + 1) / 2)  # rescale to 0–1 for imshow()\n        if(y_batch[index]==1):\n            classt='FAKE'\n        else:\n            classt='REAL'\n        plt.title(f\"Class: {classt}\")\n        plt.axis(\"off\")\n\nplt.show()","metadata":{"id":"ZL3c3i4opKz9","outputId":"38847d8d-8822-41a3-cfb2-27479aa5debe","execution":{"iopub.status.busy":"2023-04-07T16:32:14.918453Z","iopub.execute_input":"2023-04-07T16:32:14.918878Z","iopub.status.idle":"2023-04-07T16:32:16.292175Z","shell.execute_reply.started":"2023-04-07T16:32:14.918836Z","shell.execute_reply":"2023-04-07T16:32:16.291167Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_augmentation = tf.keras.Sequential([\n    tf.keras.layers.RandomFlip(mode=\"horizontal\", seed=42),\n    tf.keras.layers.RandomRotation(factor=0.05, seed=42),\n    tf.keras.layers.RandomContrast(factor=0.2, seed=42)\n])","metadata":{"id":"Ib0cA8Y1pKz9","execution":{"iopub.status.busy":"2023-04-07T16:32:16.293601Z","iopub.execute_input":"2023-04-07T16:32:16.295083Z","iopub.status.idle":"2023-04-07T16:32:16.313618Z","shell.execute_reply.started":"2023-04-07T16:32:16.29504Z","shell.execute_reply":"2023-04-07T16:32:16.312374Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extra code – displays the same first 9 images, after augmentation\n\nplt.figure(figsize=(12, 12))\nfor X_batch, y_batch in valid_set.take(1):\n    X_batch_augmented = data_augmentation(X_batch, training=True)\n    for index in range(9):\n        plt.subplot(3, 3, index + 1)\n        # We must rescale the images to the 0-1 range for imshow(), and also\n        # clip the result to that range, because data augmentation may\n        # make some values go out of bounds (e.g., RandomContrast in this case).\n        plt.imshow(np.clip((X_batch_augmented[index] + 1) / 2, 0, 1))\n        if(y_batch[index]==1):\n            classt='FAKE'\n        else:\n            classt='REAL'\n        plt.title(f\"Class: {classt}\")\n        plt.axis(\"off\")\n\nplt.show()","metadata":{"id":"w6GH5_vupKz-","outputId":"eeb2c924-2f4f-4aa1-bea9-951bebef4bf0","execution":{"iopub.status.busy":"2023-04-07T16:32:16.315735Z","iopub.execute_input":"2023-04-07T16:32:16.31666Z","iopub.status.idle":"2023-04-07T16:32:19.889764Z","shell.execute_reply.started":"2023-04-07T16:32:16.316617Z","shell.execute_reply":"2023-04-07T16:32:19.888868Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Now let's load the pretrained model, without its top layers, and replace them with our own task","metadata":{"id":"kNL9AOsDpKz-"}},{"cell_type":"code","source":"tf.random.set_seed(42)  # extra code – ensures reproducibility\nbase_model = tf.keras.applications.xception.Xception(weights=\"imagenet\",\n                                                     include_top=False)\navg = tf.keras.layers.GlobalAveragePooling2D()(base_model.output)\noutput = tf.keras.layers.Dense(1, activation=\"sigmoid\")(avg)\nmodel = tf.keras.Model(inputs=base_model.input, outputs=output)","metadata":{"id":"lRyCgvaKpKz-","outputId":"a825e173-8b1d-4217-a1c4-5491b49c3e82","execution":{"iopub.status.busy":"2023-04-07T16:32:19.891387Z","iopub.execute_input":"2023-04-07T16:32:19.891979Z","iopub.status.idle":"2023-04-07T16:32:22.087329Z","shell.execute_reply.started":"2023-04-07T16:32:19.891939Z","shell.execute_reply":"2023-04-07T16:32:22.086181Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for layer in base_model.layers:\n    layer.trainable = False","metadata":{"id":"KBlyG6ElpKz-","execution":{"iopub.status.busy":"2023-04-07T16:32:22.089079Z","iopub.execute_input":"2023-04-07T16:32:22.089502Z","iopub.status.idle":"2023-04-07T16:32:22.100396Z","shell.execute_reply.started":"2023-04-07T16:32:22.089454Z","shell.execute_reply":"2023-04-07T16:32:22.099065Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer = tf.keras.optimizers.SGD(learning_rate=0.1, momentum=0.9)\nmodel.compile(loss=\"binary_crossentropy\", optimizer=optimizer,\n              metrics=[\"accuracy\"])\nhistory = model.fit(train_set, validation_data=valid_set, epochs=3)","metadata":{"id":"GGxK2yPcpKz-","outputId":"6b64214a-e104-4b6c-9b7a-3388fc9aa15f","execution":{"iopub.status.busy":"2023-04-07T16:32:22.102691Z","iopub.execute_input":"2023-04-07T16:32:22.103541Z","iopub.status.idle":"2023-04-07T16:35:52.855849Z","shell.execute_reply.started":"2023-04-07T16:32:22.103492Z","shell.execute_reply":"2023-04-07T16:35:52.854693Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for indices in zip(range(33), range(33, 66), range(66, 99), range(99, 132)):\n    for idx in indices:\n        print(f\"{idx:3}: {base_model.layers[idx].name:22}\", end=\"\")\n    print()","metadata":{"id":"GvGMiJMLpKz-","outputId":"91f2c96c-c058-45e0-e428-66fa6076ad56","execution":{"iopub.status.busy":"2023-04-07T16:35:52.859602Z","iopub.execute_input":"2023-04-07T16:35:52.859973Z","iopub.status.idle":"2023-04-07T16:35:52.893636Z","shell.execute_reply.started":"2023-04-07T16:35:52.85994Z","shell.execute_reply":"2023-04-07T16:35:52.892489Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(test_set)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T16:35:52.895053Z","iopub.execute_input":"2023-04-07T16:35:52.895707Z","iopub.status.idle":"2023-04-07T16:36:13.783148Z","shell.execute_reply.started":"2023-04-07T16:35:52.895663Z","shell.execute_reply":"2023-04-07T16:36:13.781997Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for layer in base_model.layers[56:]:\n    layer.trainable = True\n\noptimizer = tf.keras.optimizers.SGD(learning_rate=0.01, momentum=0.9)\nmodel.compile(loss=\"binary_crossentropy\", optimizer=optimizer,\n              metrics=[\"accuracy\"])\nhistory = model.fit(train_set, validation_data=valid_set, epochs=20)","metadata":{"id":"GEUNGlhvpKz_","outputId":"c622a91d-f634-4443-b87e-8d46defdb578","execution":{"iopub.status.busy":"2023-04-07T16:36:13.784711Z","iopub.execute_input":"2023-04-07T16:36:13.785552Z","iopub.status.idle":"2023-04-07T17:14:00.840426Z","shell.execute_reply.started":"2023-04-07T16:36:13.785503Z","shell.execute_reply":"2023-04-07T17:14:00.83929Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot model performance\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs_range = range(1, len(history.epoch) + 1)\n\nplt.figure(figsize=(15,5))\n\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Train Set')\nplt.plot(epochs_range, val_acc, label='Val Set')\nplt.legend(loc=\"best\")\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.title('Model Accuracy')\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Train Set')\nplt.plot(epochs_range, val_loss, label='Val Set')\nplt.legend(loc=\"best\")\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.title('Model Loss')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T17:14:00.842728Z","iopub.execute_input":"2023-04-07T17:14:00.843092Z","iopub.status.idle":"2023-04-07T17:14:01.359358Z","shell.execute_reply.started":"2023-04-07T17:14:00.843058Z","shell.execute_reply":"2023-04-07T17:14:01.35829Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(test_set)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T17:14:01.361338Z","iopub.execute_input":"2023-04-07T17:14:01.361669Z","iopub.status.idle":"2023-04-07T17:14:15.381659Z","shell.execute_reply.started":"2023-04-07T17:14:01.361639Z","shell.execute_reply":"2023-04-07T17:14:15.380474Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.save('xception_deepfake_image.h5')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T17:14:15.383355Z","iopub.execute_input":"2023-04-07T17:14:15.384313Z","iopub.status.idle":"2023-04-07T17:14:15.966456Z","shell.execute_reply.started":"2023-04-07T17:14:15.384266Z","shell.execute_reply":"2023-04-07T17:14:15.964941Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Add Explainability to the model","metadata":{}},{"cell_type":"markdown","source":"Lets try to interpret the trained model on how it finds a image FAKE","metadata":{}},{"cell_type":"code","source":"!pip install lime","metadata":{"execution":{"iopub.status.busy":"2023-04-07T17:14:15.968393Z","iopub.execute_input":"2023-04-07T17:14:15.96874Z","iopub.status.idle":"2023-04-07T17:14:30.014127Z","shell.execute_reply.started":"2023-04-07T17:14:15.968705Z","shell.execute_reply":"2023-04-07T17:14:30.012695Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from lime import lime_image\nimport matplotlib.pyplot as plt\n\nexplainer = lime_image.LimeImageExplainer()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T17:14:30.017301Z","iopub.execute_input":"2023-04-07T17:14:30.017683Z","iopub.status.idle":"2023-04-07T17:14:31.506996Z","shell.execute_reply.started":"2023-04-07T17:14:30.017642Z","shell.execute_reply":"2023-04-07T17:14:31.505819Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(12, 12))\nfor X_batch, y_batch in test_set.take(1):\n    X_batch_augmented = data_augmentation(X_batch, training=True)\n    for index in range(9):\n        plt.subplot(3, 3, index + 1)\n        plt.imshow(np.clip((X_batch_augmented[index] + 1) / 2, 0, 1))\n        if(y_batch[index]==1):\n            classt='FAKE'\n        else:\n            classt='REAL'\n        plt.title(f\"Class: {classt}\")\n        plt.axis(\"off\")\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-07T17:14:31.510946Z","iopub.execute_input":"2023-04-07T17:14:31.512932Z","iopub.status.idle":"2023-04-07T17:14:34.293768Z","shell.execute_reply.started":"2023-04-07T17:14:31.51287Z","shell.execute_reply":"2023-04-07T17:14:34.292793Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_data = X_test[2,:,:,:]\nfor X_batch, y_batch in test_set.take(1):\n    test_data = data_augmentation(X_batch, training=True)\n\ntest_data = np.array(test_data[2,:,:,:])\ntest_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-04-07T17:14:34.29554Z","iopub.execute_input":"2023-04-07T17:14:34.296774Z","iopub.status.idle":"2023-04-07T17:14:36.190837Z","shell.execute_reply.started":"2023-04-07T17:14:34.29673Z","shell.execute_reply":"2023-04-07T17:14:36.189758Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"explanation = explainer.explain_instance(test_data.astype('double'), model.predict,  \n                                         top_labels=3, hide_color=0, num_samples=1000)","metadata":{"execution":{"iopub.status.busy":"2023-04-07T17:14:36.200499Z","iopub.execute_input":"2023-04-07T17:14:36.200825Z","iopub.status.idle":"2023-04-07T17:14:53.409167Z","shell.execute_reply.started":"2023-04-07T17:14:36.200777Z","shell.execute_reply":"2023-04-07T17:14:53.407627Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from skimage.segmentation import mark_boundaries\n\ntemp_1, mask_1 = explanation.get_image_and_mask(explanation.top_labels[0], positive_only=True, num_features=5, hide_rest=True)\ntemp_2, mask_2 = explanation.get_image_and_mask(explanation.top_labels[0], positive_only=False, num_features=10, hide_rest=False)\n\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(15,15))\nax1.imshow(mark_boundaries(temp_1, mask_1))\nax2.imshow(mark_boundaries(temp_2, mask_2))\nax1.axis('off')\nax2.axis('off')","metadata":{"execution":{"iopub.status.busy":"2023-04-07T17:14:53.413685Z","iopub.execute_input":"2023-04-07T17:14:53.416923Z","iopub.status.idle":"2023-04-07T17:14:53.878959Z","shell.execute_reply.started":"2023-04-07T17:14:53.416857Z","shell.execute_reply":"2023-04-07T17:14:53.877835Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fake video detection\n","metadata":{}},{"cell_type":"code","source":"from tensorflow import keras\n#from imutils import paths\n\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport pandas as pd\nimport numpy as np\nimport imageio\nimport cv2\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-04-07T18:05:04.801408Z","iopub.execute_input":"2023-04-07T18:05:04.802448Z","iopub.status.idle":"2023-04-07T18:05:04.948612Z","shell.execute_reply.started":"2023-04-07T18:05:04.802397Z","shell.execute_reply":"2023-04-07T18:05:04.947573Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Data Visualisation","metadata":{"execution":{"iopub.status.busy":"2023-04-07T18:05:05.236675Z","iopub.execute_input":"2023-04-07T18:05:05.237285Z","iopub.status.idle":"2023-04-07T18:05:05.2421Z","shell.execute_reply.started":"2023-04-07T18:05:05.237241Z","shell.execute_reply":"2023-04-07T18:05:05.240881Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"DATA_FOLDER = '../input/deepfake-detection-challenge'\nTRAIN_SAMPLE_FOLDER = 'train_sample_videos'\nTEST_FOLDER = 'test_videos'\n\nprint(f\"Train samples: {len(os.listdir(os.path.join(DATA_FOLDER, TRAIN_SAMPLE_FOLDER)))}\")\nprint(f\"Test samples: {len(os.listdir(os.path.join(DATA_FOLDER, TEST_FOLDER)))}\")\n\ntrain_sample_metadata = pd.read_json('../input/deepfake-detection-challenge/train_sample_videos/metadata.json').T\ntrain_sample_metadata.head()\n\ntrain_sample_metadata.groupby('label')['label'].count().plot(figsize=(15, 5), kind='bar', title='Distribution of Labels in the Training Set')\nplt.show()\n\ntrain_sample_metadata.shape","metadata":{"execution":{"iopub.status.busy":"2025-05-27T18:38:42.162269Z","iopub.execute_input":"2025-05-27T18:38:42.162906Z","iopub.status.idle":"2025-05-27T18:38:42.460753Z","shell.execute_reply.started":"2025-05-27T18:38:42.162881Z","shell.execute_reply":"2025-05-27T18:38:42.459999Z"},"trusted":true},"outputs":[],"execution_count":null}]}