{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":924245,"sourceType":"datasetVersion","datasetId":464091}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Deep Fake Detection using CNN and RNN","metadata":{"id":"Q6tasuafvT2O"}},{"cell_type":"markdown","source":"## Importing Required libraries","metadata":{"id":"dFXIv9qNpKzt","tags":[]}},{"cell_type":"code","source":"import sys\nimport sklearn\nimport tensorflow as tf\n\nimport cv2\nimport pandas as pd\nimport numpy as np\n\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nfrom matplotlib import pyplot as plt","metadata":{"id":"TFSU3FCOpKzu","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:11:49.530489Z","iopub.execute_input":"2024-11-16T17:11:49.530799Z","iopub.status.idle":"2024-11-16T17:12:02.557799Z","shell.execute_reply.started":"2024-11-16T17:11:49.530765Z","shell.execute_reply":"2024-11-16T17:12:02.556818Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.test.is_gpu_available()\nstrategy = tf.distribute.MirroredStrategy()\nprint('DEVICES AVAILABLE: {}'.format(strategy.num_replicas_in_sync))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:12:02.56128Z","iopub.execute_input":"2024-11-16T17:12:02.561816Z","iopub.status.idle":"2024-11-16T17:12:03.607204Z","shell.execute_reply.started":"2024-11-16T17:12:02.561778Z","shell.execute_reply":"2024-11-16T17:12:03.605987Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.__version__","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:12:03.608593Z","iopub.execute_input":"2024-11-16T17:12:03.608946Z","iopub.status.idle":"2024-11-16T17:12:03.617974Z","shell.execute_reply.started":"2024-11-16T17:12:03.608907Z","shell.execute_reply":"2024-11-16T17:12:03.61674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.rc('font', size=14)\nplt.rc('axes', labelsize=14, titlesize=14)\nplt.rc('legend', fontsize=14)\nplt.rc('xtick', labelsize=10)\nplt.rc('ytick', labelsize=10)","metadata":{"id":"8d4TH3NbpKzx","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:12:03.621057Z","iopub.execute_input":"2024-11-16T17:12:03.62152Z","iopub.status.idle":"2024-11-16T17:12:03.628174Z","shell.execute_reply.started":"2024-11-16T17:12:03.621472Z","shell.execute_reply":"2024-11-16T17:12:03.626977Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Visualisation","metadata":{"id":"NL3Ht4wC9b3n"}},{"cell_type":"code","source":"import os\n\ndef get_data():\n    return pd.read_csv('../input/deepfake-faces/metadata.csv')","metadata":{"id":"jfv9PxSB4tM8","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:12:03.629635Z","iopub.execute_input":"2024-11-16T17:12:03.629988Z","iopub.status.idle":"2024-11-16T17:12:03.639244Z","shell.execute_reply.started":"2024-11-16T17:12:03.629952Z","shell.execute_reply":"2024-11-16T17:12:03.638374Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta=get_data()\nmeta.head()","metadata":{"id":"tDW7BRph9ehF","outputId":"97de18b5-0a37-4302-8804-8a16a7d2ed2f","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:12:03.640442Z","iopub.execute_input":"2024-11-16T17:12:03.64077Z","iopub.status.idle":"2024-11-16T17:12:03.822705Z","shell.execute_reply.started":"2024-11-16T17:12:03.640739Z","shell.execute_reply":"2024-11-16T17:12:03.821538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"meta.shape","metadata":{"id":"n7FSdDifbZxn","outputId":"5451a127-405a-4c0b-a197-c920b796adbb","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:12:03.824035Z","iopub.execute_input":"2024-11-16T17:12:03.824487Z","iopub.status.idle":"2024-11-16T17:12:03.83143Z","shell.execute_reply.started":"2024-11-16T17:12:03.824439Z","shell.execute_reply":"2024-11-16T17:12:03.830361Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(meta[meta.label=='FAKE']),len(meta[meta.label=='REAL'])","metadata":{"id":"_FJcz2IthxVG","outputId":"274c3f65-7acb-4f99-8aa9-a5b2a23bf06a","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:12:08.036644Z","iopub.execute_input":"2024-11-16T17:12:08.037294Z","iopub.status.idle":"2024-11-16T17:12:08.085129Z","shell.execute_reply.started":"2024-11-16T17:12:08.037252Z","shell.execute_reply":"2024-11-16T17:12:08.084247Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"real_df = meta[meta[\"label\"] == \"REAL\"]\nfake_df = meta[meta[\"label\"] == \"FAKE\"]\nsample_size = 15000\nfake_df = fake_df.sample(sample_size, random_state=42)\nsample_meta = pd.concat([real_df, fake_df])","metadata":{"id":"IgMfzY-PjjtH","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:12:11.307675Z","iopub.execute_input":"2024-11-16T17:12:11.308287Z","iopub.status.idle":"2024-11-16T17:12:11.361952Z","shell.execute_reply.started":"2024-11-16T17:12:11.308246Z","shell.execute_reply":"2024-11-16T17:12:11.361208Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"As mentioned instead of using 95k images we will only use 16000 images.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nTrain_set, Test_set = train_test_split(sample_meta,test_size=0.2,random_state=42,stratify=sample_meta['label'])\nTrain_set, Val_set  = train_test_split(Train_set,test_size=0.15,random_state=42,stratify=Train_set['label'])","metadata":{"id":"5eB86S6K-T5Z","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:12:12.918297Z","iopub.execute_input":"2024-11-16T17:12:12.91868Z","iopub.status.idle":"2024-11-16T17:12:13.097802Z","shell.execute_reply.started":"2024-11-16T17:12:12.918641Z","shell.execute_reply":"2024-11-16T17:12:13.097025Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"Train_set.shape,Val_set.shape,Test_set.shape","metadata":{"id":"8p-TONijb4qA","outputId":"56d0b529-9d81-4019-d8fa-618c8cdba90f","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:12:15.298667Z","iopub.execute_input":"2024-11-16T17:12:15.299039Z","iopub.status.idle":"2024-11-16T17:12:15.305656Z","shell.execute_reply.started":"2024-11-16T17:12:15.299003Z","shell.execute_reply":"2024-11-16T17:12:15.304586Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y = dict()\n\ny[0] = []\ny[1] = []\n\nfor set_name in (np.array(Train_set['label']), np.array(Val_set['label']), np.array(Test_set['label'])):\n    y[0].append(np.sum(set_name == 'REAL'))\n    y[1].append(np.sum(set_name == 'FAKE'))\n\ntrace0 = go.Bar(\n    x=['Train Set', 'Validation Set', 'Test Set'],\n    y=y[0],\n    name='REAL',\n    marker=dict(color='#33cc33'),\n    opacity=0.7\n)\ntrace1 = go.Bar(\n    x=['Train Set', 'Validation Set', 'Test Set'],\n    y=y[1],\n    name='FAKE',\n    marker=dict(color='#ff3300'),\n    opacity=0.7\n)\n\ndata = [trace0, trace1]\nlayout = go.Layout(\n    title='Count of classes in each set',\n    xaxis={'title': 'Set'},\n    yaxis={'title': 'Count'}\n)\n\nfig = go.Figure(data, layout)\niplot(fig)","metadata":{"id":"hzNGtCWd-mTk","outputId":"5178c3ed-cbba-4f99-99d0-26bde11a5dab","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:12:18.263896Z","iopub.execute_input":"2024-11-16T17:12:18.264285Z","iopub.status.idle":"2024-11-16T17:12:19.61996Z","shell.execute_reply.started":"2024-11-16T17:12:18.264247Z","shell.execute_reply":"2024-11-16T17:12:19.619124Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"The original image dataset were biased with more fake images than real since we are taking a sample of it its better to take equal proportion of real and fake images.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,15))\nfor cur,i in enumerate(Train_set.index[25:50]):\n    plt.subplot(5,5,cur+1)\n    plt.xticks([])\n    plt.yticks([])\n    plt.grid(False)\n    \n    plt.imshow(cv2.imread('../input/deepfake-faces/faces_224/'+Train_set.loc[i,'videoname'][:-4]+'.jpg'))\n    \n    if(Train_set.loc[i,'label']=='FAKE'):\n        plt.xlabel('FAKE Image')\n    else:\n        plt.xlabel('REAL Image')\n        \nplt.show()","metadata":{"id":"VR7Uly2fcUYi","outputId":"c1f47a82-ef4f-4bcd-b51c-d4738142fc0f","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:12:25.292808Z","iopub.execute_input":"2024-11-16T17:12:25.293542Z","iopub.status.idle":"2024-11-16T17:12:27.406257Z","shell.execute_reply.started":"2024-11-16T17:12:25.293498Z","shell.execute_reply":"2024-11-16T17:12:27.405057Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Base Model","metadata":{"id":"dOvN_divkl-N"}},{"cell_type":"markdown","source":"### Custom CNN Architecture","metadata":{"id":"oid44Xx-pKz6"}},{"cell_type":"code","source":"def retreive_dataset(set_name):\n    images,labels=[],[]\n    for (img, imclass) in zip(set_name['videoname'], set_name['label']):\n        images.append(cv2.imread('../input/deepfake-faces/faces_224/'+img[:-4]+'.jpg'))\n        if(imclass=='FAKE'):\n            labels.append(1)\n        else:\n            labels.append(0)\n    \n    return np.array(images),np.array(labels)","metadata":{"id":"Hz0ZdQ_fgHhG","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:13:45.624404Z","iopub.execute_input":"2024-11-16T17:13:45.625233Z","iopub.status.idle":"2024-11-16T17:13:45.631341Z","shell.execute_reply.started":"2024-11-16T17:13:45.625174Z","shell.execute_reply":"2024-11-16T17:13:45.630347Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train,y_train=retreive_dataset(Train_set)\nX_val,y_val=retreive_dataset(Val_set)\nX_test,y_test=retreive_dataset(Test_set)","metadata":{"id":"zeAGRcAbguKU","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:13:46.113256Z","iopub.execute_input":"2024-11-16T17:13:46.113601Z","iopub.status.idle":"2024-11-16T17:18:00.645945Z","shell.execute_reply.started":"2024-11-16T17:13:46.113567Z","shell.execute_reply":"2024-11-16T17:18:00.644727Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Pretrained Models for Transfer Learning","metadata":{"id":"hqxnSBJ3pKz8"}},{"cell_type":"markdown","source":"using Xception model for fine-tuning ","metadata":{}},{"cell_type":"code","source":"train_set_raw=tf.data.Dataset.from_tensor_slices((X_train,y_train))\nvalid_set_raw=tf.data.Dataset.from_tensor_slices((X_val,y_val))\ntest_set_raw=tf.data.Dataset.from_tensor_slices((X_test,y_test))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:18:10.804429Z","iopub.execute_input":"2024-11-16T17:18:10.804737Z","iopub.status.idle":"2024-11-16T17:18:21.188054Z","shell.execute_reply.started":"2024-11-16T17:18:10.804705Z","shell.execute_reply":"2024-11-16T17:18:21.187264Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.keras.backend.clear_session()  # extra code – resets layer name counter\n\nbatch_size_per_replica = 32\nbatch_size = batch_size_per_replica\npreprocess = tf.keras.applications.xception.preprocess_input\ntrain_set = train_set_raw.map(lambda X, y: (preprocess(tf.cast(X, tf.float32)), y))\ntrain_set = train_set.shuffle(1000, seed=42).batch(batch_size).prefetch(1)\nvalid_set = valid_set_raw.map(lambda X, y: (preprocess(tf.cast(X, tf.float32)), y)).batch(batch_size)\ntest_set = test_set_raw.map(lambda X, y: (preprocess(tf.cast(X, tf.float32)), y)).batch(batch_size)","metadata":{"id":"Bnz0n9XApKz9","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:18:21.190003Z","iopub.execute_input":"2024-11-16T17:18:21.190324Z","iopub.status.idle":"2024-11-16T17:18:22.989794Z","shell.execute_reply.started":"2024-11-16T17:18:21.19029Z","shell.execute_reply":"2024-11-16T17:18:22.989017Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extra code – displays the first 9 images in the first batch of valid_set\n\nplt.figure(figsize=(12, 12))\nfor X_batch, y_batch in valid_set.take(1):\n    for index in range(9):\n        plt.subplot(3, 3, index + 1)\n        plt.imshow((X_batch[index] + 1) / 2)  # rescale to 0–1 for imshow()\n        if(y_batch[index]==1):\n            classt='FAKE'\n        else:\n            classt='REAL'\n        plt.title(f\"Class: {classt}\")\n        plt.axis(\"off\")\n\nplt.show()","metadata":{"id":"ZL3c3i4opKz9","outputId":"38847d8d-8822-41a3-cfb2-27479aa5debe","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:18:22.990827Z","iopub.execute_input":"2024-11-16T17:18:22.991136Z","iopub.status.idle":"2024-11-16T17:18:24.39996Z","shell.execute_reply.started":"2024-11-16T17:18:22.991089Z","shell.execute_reply":"2024-11-16T17:18:24.399093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_augmentation = tf.keras.Sequential([\n    tf.keras.layers.RandomFlip(mode=\"horizontal\", seed=42),\n    tf.keras.layers.RandomRotation(factor=0.05, seed=42),\n    tf.keras.layers.RandomContrast(factor=0.2, seed=42)\n])","metadata":{"id":"Ib0cA8Y1pKz9","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:18:24.401184Z","iopub.execute_input":"2024-11-16T17:18:24.401492Z","iopub.status.idle":"2024-11-16T17:18:24.418972Z","shell.execute_reply.started":"2024-11-16T17:18:24.401458Z","shell.execute_reply":"2024-11-16T17:18:24.418053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# extra code – displays the same first 9 images, after augmentation\n\nplt.figure(figsize=(12, 12))\nfor X_batch, y_batch in valid_set.take(1):\n    X_batch_augmented = data_augmentation(X_batch, training=True)\n    for index in range(9):\n        plt.subplot(3, 3, index + 1)\n        # We must rescale the images to the 0-1 range for imshow(), and also\n        # clip the result to that range, because data augmentation may\n        # make some values go out of bounds (e.g., RandomContrast in this case).\n        plt.imshow(np.clip((X_batch_augmented[index] + 1) / 2, 0, 1))\n        if(y_batch[index]==1):\n            classt='FAKE'\n        else:\n            classt='REAL'\n        plt.title(f\"Class: {classt}\")\n        plt.axis(\"off\")\n\nplt.show()","metadata":{"id":"w6GH5_vupKz-","outputId":"eeb2c924-2f4f-4aa1-bea9-951bebef4bf0","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:18:24.420347Z","iopub.execute_input":"2024-11-16T17:18:24.420979Z","iopub.status.idle":"2024-11-16T17:18:27.126683Z","shell.execute_reply.started":"2024-11-16T17:18:24.420933Z","shell.execute_reply":"2024-11-16T17:18:27.125722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tf.random.set_seed(42)  # extra code – ensures reproducibility\nbase_model = tf.keras.applications.xception.Xception(weights=\"imagenet\",\n                                                     include_top=False)\navg = tf.keras.layers.GlobalAveragePooling2D()(base_model.output)\noutput = tf.keras.layers.Dense(1, activation=\"sigmoid\")(avg)\nmodel = tf.keras.Model(inputs=base_model.input, outputs=output)","metadata":{"id":"lRyCgvaKpKz-","outputId":"a825e173-8b1d-4217-a1c4-5491b49c3e82","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:18:27.128477Z","iopub.execute_input":"2024-11-16T17:18:27.128858Z","iopub.status.idle":"2024-11-16T17:18:28.783438Z","shell.execute_reply.started":"2024-11-16T17:18:27.12879Z","shell.execute_reply":"2024-11-16T17:18:28.782629Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**After Fine tuning The accuracy comes about ~64%**","metadata":{}},{"cell_type":"markdown","source":"Now that the weights of our new top layers are not too bad, we can make the top part of the base model trainable again, and continue training, but with a lower learning rate:","metadata":{"id":"L_bEwL8KpKz_"}},{"cell_type":"code","source":"for layer in base_model.layers[56:]:\n    layer.trainable = True\n\noptimizer = tf.keras.optimizers.SGD(learning_rate=0.01, momentum=0.9)\nmodel.compile(loss=\"binary_crossentropy\", optimizer=optimizer,\n              metrics=[\"accuracy\"])\nhistory = model.fit(train_set, validation_data=valid_set, epochs=10)","metadata":{"id":"GEUNGlhvpKz_","outputId":"c622a91d-f634-4443-b87e-8d46defdb578","trusted":true,"execution":{"iopub.status.busy":"2024-11-16T17:18:28.784594Z","iopub.execute_input":"2024-11-16T17:18:28.784919Z","iopub.status.idle":"2024-11-16T18:07:18.611674Z","shell.execute_reply.started":"2024-11-16T17:18:28.784875Z","shell.execute_reply":"2024-11-16T18:07:18.610819Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot model performance\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs_range = range(1, len(history.epoch) + 1)\n\nplt.figure(figsize=(15,5))\n\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Train Set')\nplt.plot(epochs_range, val_acc, label='Val Set')\nplt.legend(loc=\"best\")\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.title('Model Accuracy')\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Train Set')\nplt.plot(epochs_range, val_loss, label='Val Set')\nplt.legend(loc=\"best\")\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.title('Model Loss')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T18:07:18.623742Z","iopub.execute_input":"2024-11-16T18:07:18.624074Z","iopub.status.idle":"2024-11-16T18:07:19.237595Z","shell.execute_reply.started":"2024-11-16T18:07:18.624022Z","shell.execute_reply":"2024-11-16T18:07:19.236647Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(test_set)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-16T18:07:19.2388Z","iopub.execute_input":"2024-11-16T18:07:19.239135Z","iopub.status.idle":"2024-11-16T18:07:47.43282Z","shell.execute_reply.started":"2024-11-16T18:07:19.239074Z","shell.execute_reply":"2024-11-16T18:07:47.431886Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reference","metadata":{}},{"cell_type":"markdown","source":"https://keras.io/examples/vision/video_classification/\n\nhttps://www.kaggle.com/code/gpreda/deepfake-starter-kit\n\nhttps://www.kaggle.com/code/robikscube/kaggle-deepfake-detection-introduction\n\nhttps://www.kaggle.com/code/humananalog/binary-image-classifier-training-demo\n\nhttps://www.kaggle.com/datasets/dagnelies/deepfake-faces\n\nhttps://www.kaggle.com/code/gautam20bce1227/fake-detection-on-images","metadata":{}}]}