{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Title","metadata":{"id":"Q6tasuafvT2O"}},{"cell_type":"markdown","source":"Deep Fake Image and Video Detection using CNN's and RNN's","metadata":{"id":"LvJBeHv3vfE_"}},{"cell_type":"markdown","source":"# DeepFake Image detection","metadata":{"id":"vGkKRV1kY8Z-"}},{"cell_type":"markdown","source":"## About the Dataset","metadata":{"id":"ZLLkoLT1wsPL"}},{"cell_type":"markdown","source":"This dataset contains faces extracted from deepfake-detection-challenge. All images were of size 224x224. \n\nDue to memory issue we will only use a sample of the entire dataset for prediction.","metadata":{"id":"y_3tT_2lwv-v"}},{"cell_type":"markdown","source":"## Importing Required libraries","metadata":{"id":"dFXIv9qNpKzt","tags":[]}},{"cell_type":"code","source":"!pip install -U --upgrade tensorflow","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:47:38.784345Z","iopub.execute_input":"2022-07-09T09:47:38.784663Z","iopub.status.idle":"2022-07-09T09:49:22.605841Z","shell.execute_reply.started":"2022-07-09T09:47:38.784606Z","shell.execute_reply":"2022-07-09T09:49:22.60461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nimport sklearn\nimport tensorflow as tf\n\nimport cv2\nimport pandas as pd\nimport numpy as np\n\nimport plotly.graph_objs as go\nfrom plotly.offline import iplot\nfrom matplotlib import pyplot as plt","metadata":{"id":"TFSU3FCOpKzu","execution":{"iopub.status.busy":"2023-10-23T19:37:28.078898Z","iopub.execute_input":"2023-10-23T19:37:28.079230Z","iopub.status.idle":"2023-10-23T19:37:34.512523Z","shell.execute_reply.started":"2023-10-23T19:37:28.079167Z","shell.execute_reply":"2023-10-23T19:37:34.511616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.test.is_gpu_available()","metadata":{"execution":{"iopub.status.busy":"2023-10-23T19:37:34.514093Z","iopub.execute_input":"2023-10-23T19:37:34.514303Z","iopub.status.idle":"2023-10-23T19:37:36.581967Z","shell.execute_reply.started":"2023-10-23T19:37:34.514271Z","shell.execute_reply":"2023-10-23T19:37:36.581166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.__version__","metadata":{"execution":{"iopub.status.busy":"2023-10-23T19:37:36.583373Z","iopub.execute_input":"2023-10-23T19:37:36.583833Z","iopub.status.idle":"2023-10-23T19:37:36.592830Z","shell.execute_reply.started":"2023-10-23T19:37:36.583603Z","shell.execute_reply":"2023-10-23T19:37:36.591877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.rc('font', size=14)\nplt.rc('axes', labelsize=14, titlesize=14)\nplt.rc('legend', fontsize=14)\nplt.rc('xtick', labelsize=10)\nplt.rc('ytick', labelsize=10)","metadata":{"id":"8d4TH3NbpKzx","execution":{"iopub.status.busy":"2023-10-23T19:41:40.818749Z","iopub.execute_input":"2023-10-23T19:41:40.819108Z","iopub.status.idle":"2023-10-23T19:41:40.824845Z","shell.execute_reply.started":"2023-10-23T19:41:40.819049Z","shell.execute_reply":"2023-10-23T19:41:40.824001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Visualisation","metadata":{"id":"NL3Ht4wC9b3n"}},{"cell_type":"code","source":"import os\n\ndef get_data():\n    return pd.read_csv('../input/deepfake-faces/metadata.csv')","metadata":{"id":"jfv9PxSB4tM8","execution":{"iopub.status.busy":"2023-10-23T19:42:01.408371Z","iopub.execute_input":"2023-10-23T19:42:01.408707Z","iopub.status.idle":"2023-10-23T19:42:01.412776Z","shell.execute_reply.started":"2023-10-23T19:42:01.408642Z","shell.execute_reply":"2023-10-23T19:42:01.411911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta=get_data()\nmeta.head()","metadata":{"id":"tDW7BRph9ehF","outputId":"97de18b5-0a37-4302-8804-8a16a7d2ed2f","execution":{"iopub.status.busy":"2023-10-23T19:42:01.593103Z","iopub.execute_input":"2023-10-23T19:42:01.593425Z","iopub.status.idle":"2023-10-23T19:42:01.794739Z","shell.execute_reply.started":"2023-10-23T19:42:01.593369Z","shell.execute_reply":"2023-10-23T19:42:01.794076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta.shape","metadata":{"id":"n7FSdDifbZxn","outputId":"5451a127-405a-4c0b-a197-c920b796adbb","execution":{"iopub.status.busy":"2023-10-23T19:42:20.978763Z","iopub.execute_input":"2023-10-23T19:42:20.979102Z","iopub.status.idle":"2023-10-23T19:42:20.984241Z","shell.execute_reply.started":"2023-10-23T19:42:20.979047Z","shell.execute_reply":"2023-10-23T19:42:20.983327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(meta[meta.label=='FAKE']),len(meta[meta.label=='REAL'])","metadata":{"id":"_FJcz2IthxVG","outputId":"274c3f65-7acb-4f99-8aa9-a5b2a23bf06a","execution":{"iopub.status.busy":"2023-10-23T19:42:21.271933Z","iopub.execute_input":"2023-10-23T19:42:21.272240Z","iopub.status.idle":"2023-10-23T19:42:21.316333Z","shell.execute_reply.started":"2023-10-23T19:42:21.272186Z","shell.execute_reply":"2023-10-23T19:42:21.315440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_df = meta[meta[\"label\"] == \"REAL\"]\nfake_df = meta[meta[\"label\"] == \"FAKE\"]\nsample_size = 8000\n\nreal_df = real_df.sample(sample_size, random_state=42)\nfake_df = fake_df.sample(sample_size, random_state=42)\n\nsample_meta = pd.concat([real_df, fake_df])","metadata":{"id":"IgMfzY-PjjtH","execution":{"iopub.status.busy":"2023-10-23T19:42:24.853612Z","iopub.execute_input":"2023-10-23T19:42:24.853939Z","iopub.status.idle":"2023-10-23T19:42:24.907854Z","shell.execute_reply.started":"2023-10-23T19:42:24.853891Z","shell.execute_reply":"2023-10-23T19:42:24.907076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As mentioned instead of using 95k images we will only use 16000 images.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nTrain_set, Test_set = train_test_split(sample_meta,test_size=0.2,random_state=42,stratify=sample_meta['label'])\nTrain_set, Val_set  = train_test_split(Train_set,test_size=0.3,random_state=42,stratify=Train_set['label'])","metadata":{"id":"5eB86S6K-T5Z","execution":{"iopub.status.busy":"2023-10-23T19:42:27.735382Z","iopub.execute_input":"2023-10-23T19:42:27.735672Z","iopub.status.idle":"2023-10-23T19:42:28.316217Z","shell.execute_reply.started":"2023-10-23T19:42:27.735629Z","shell.execute_reply":"2023-10-23T19:42:28.315343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Train_set.shape,Val_set.shape,Test_set.shape","metadata":{"id":"8p-TONijb4qA","outputId":"56d0b529-9d81-4019-d8fa-618c8cdba90f","execution":{"iopub.status.busy":"2023-10-23T19:42:41.157589Z","iopub.execute_input":"2023-10-23T19:42:41.157990Z","iopub.status.idle":"2023-10-23T19:42:41.171727Z","shell.execute_reply.started":"2023-10-23T19:42:41.157930Z","shell.execute_reply":"2023-10-23T19:42:41.170754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = dict()\n\ny[0] = []\ny[1] = []\n\nfor set_name in (np.array(Train_set['label']), np.array(Val_set['label']), np.array(Test_set['label'])):\n    y[0].append(np.sum(set_name == 'REAL'))\n    y[1].append(np.sum(set_name == 'FAKE'))\n\ntrace0 = go.Bar(\n    x=['Train Set', 'Validation Set', 'Test Set'],\n    y=y[0],\n    name='REAL',\n    marker=dict(color='#33cc33'),\n    opacity=0.7\n)\ntrace1 = go.Bar(\n    x=['Train Set', 'Validation Set', 'Test Set'],\n    y=y[1],\n    name='FAKE',\n    marker=dict(color='#ff3300'),\n    opacity=0.7\n)\n\ndata = [trace0, trace1]\nlayout = go.Layout(\n    title='Count of classes in each set',\n    xaxis={'title': 'Set'},\n    yaxis={'title': 'Count'}\n)\n\nfig = go.Figure(data, layout)\niplot(fig)","metadata":{"id":"hzNGtCWd-mTk","outputId":"5178c3ed-cbba-4f99-99d0-26bde11a5dab","execution":{"iopub.status.busy":"2023-10-23T19:43:07.370457Z","iopub.execute_input":"2023-10-23T19:43:07.370746Z","iopub.status.idle":"2023-10-23T19:43:07.746832Z","shell.execute_reply.started":"2023-10-23T19:43:07.370709Z","shell.execute_reply":"2023-10-23T19:43:07.746116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The original image dataset were biased with more fake images than real since we are taking a sample of it its better to take equal proportion of real and fake images.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,15))\nfor cur,i in enumerate(Train_set.index[25:50]):\n    plt.subplot(5,5,cur+1)\n    plt.xticks([])\n    plt.yticks([])\n    plt.grid(False)\n    \n    plt.imshow(cv2.imread('../input/deepfake-faces/faces_224/'+Train_set.loc[i,'videoname'][:-4]+'.jpg'))\n    \n    if(Train_set.loc[i,'label']=='FAKE'):\n        plt.xlabel('FAKE Image')\n    else:\n        plt.xlabel('REAL Image')\n        \nplt.show()","metadata":{"id":"VR7Uly2fcUYi","outputId":"c1f47a82-ef4f-4bcd-b51c-d4738142fc0f","execution":{"iopub.status.busy":"2023-10-23T19:43:12.575544Z","iopub.execute_input":"2023-10-23T19:43:12.575889Z","iopub.status.idle":"2023-10-23T19:43:14.471342Z","shell.execute_reply.started":"2023-10-23T19:43:12.575828Z","shell.execute_reply":"2023-10-23T19:43:14.470483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modelling","metadata":{"id":"dOvN_divkl-N"}},{"cell_type":"markdown","source":"Before jumping to use pretrained model lets develop some base line model to test how our pretrained model outperforms.","metadata":{}},{"cell_type":"markdown","source":"### Custom CNN Architecture","metadata":{"id":"oid44Xx-pKz6"}},{"cell_type":"code","source":"def retreive_dataset(set_name):\n    images,labels=[],[]\n    for (img, imclass) in zip(set_name['videoname'], set_name['label']):\n        images.append(cv2.imread('../input/deepfake-faces/faces_224/'+img[:-4]+'.jpg'))\n        if(imclass=='FAKE'):\n            labels.append(1)\n        else:\n            labels.append(0)\n    \n    return np.array(images),np.array(labels)","metadata":{"id":"Hz0ZdQ_fgHhG","execution":{"iopub.status.busy":"2023-10-23T19:43:23.103999Z","iopub.execute_input":"2023-10-23T19:43:23.104329Z","iopub.status.idle":"2023-10-23T19:43:23.111027Z","shell.execute_reply.started":"2023-10-23T19:43:23.104270Z","shell.execute_reply":"2023-10-23T19:43:23.110122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train,y_train=retreive_dataset(Train_set)\nX_val,y_val=retreive_dataset(Val_set)\nX_test,y_test=retreive_dataset(Test_set)","metadata":{"id":"zeAGRcAbguKU","execution":{"iopub.status.busy":"2023-10-23T19:43:25.812763Z","iopub.execute_input":"2023-10-23T19:43:25.813137Z","iopub.status.idle":"2023-10-23T19:45:58.380594Z","shell.execute_reply.started":"2023-10-23T19:43:25.813079Z","shell.execute_reply":"2023-10-23T19:45:58.379638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from functools import partial\n\ntf.random.set_seed(42) \nDefaultConv2D = partial(tf.keras.layers.Conv2D, kernel_size=3, padding=\"same\",\n                        activation=\"relu\", kernel_initializer=\"he_normal\")\n\nmodel = tf.keras.Sequential([\n    DefaultConv2D(filters=64, kernel_size=7, input_shape=[224, 224, 3]),\n    tf.keras.layers.MaxPool2D(),\n    DefaultConv2D(filters=128),\n    DefaultConv2D(filters=128),\n    tf.keras.layers.MaxPool2D(),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(units=128, activation=\"relu\",\n                          kernel_initializer=\"he_normal\"),\n    tf.keras.layers.Dropout(0.5),\n    tf.keras.layers.Dense(units=64, activation=\"relu\",\n                          kernel_initializer=\"he_normal\"),\n    tf.keras.layers.Dropout(0.5),\n    tf.keras.layers.Dense(units=1, activation=\"sigmoid\")\n])","metadata":{"id":"34upiak4pKz6","execution":{"iopub.status.busy":"2023-10-23T19:45:58.382323Z","iopub.execute_input":"2023-10-23T19:45:58.382566Z","iopub.status.idle":"2023-10-23T19:45:58.674117Z","shell.execute_reply.started":"2023-10-23T19:45:58.382527Z","shell.execute_reply":"2023-10-23T19:45:58.673236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss=\"binary_crossentropy\", optimizer=\"nadam\",\n              metrics=[\"accuracy\"])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-10-23T19:47:34.819213Z","iopub.execute_input":"2023-10-23T19:47:34.819547Z","iopub.status.idle":"2023-10-23T19:47:34.868507Z","shell.execute_reply.started":"2023-10-23T19:47:34.819499Z","shell.execute_reply":"2023-10-23T19:47:34.867457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs=5,batch_size=64,\n                    validation_data=(X_val, y_val))","metadata":{"id":"KZbWeIBYpKz6","outputId":"deb6f56a-7b93-4241-a1bd-b210c0f2d426","execution":{"iopub.status.busy":"2023-10-23T19:47:35.008435Z","iopub.execute_input":"2023-10-23T19:47:35.008730Z","iopub.status.idle":"2023-10-23T19:49:54.183925Z","shell.execute_reply.started":"2023-10-23T19:47:35.008687Z","shell.execute_reply":"2023-10-23T19:49:54.183087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score = model.evaluate(X_test, y_test)","metadata":{"id":"6HDDr4uehast","execution":{"iopub.status.busy":"2023-10-23T19:49:58.498553Z","iopub.execute_input":"2023-10-23T19:49:58.498921Z","iopub.status.idle":"2023-10-23T19:50:02.077480Z","shell.execute_reply.started":"2023-10-23T19:49:58.498861Z","shell.execute_reply":"2023-10-23T19:50:02.076828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot model performance\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\nepochs_range = range(1, len(history.epoch) + 1)\n\nplt.figure(figsize=(15,5))\n\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Train Set')\nplt.plot(epochs_range, val_acc, label='Val Set')\nplt.legend(loc=\"best\")\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.title('Model Accuracy')\n\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Train Set')\nplt.plot(epochs_range, val_loss, label='Val Set')\nplt.legend(loc=\"best\")\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.title('Model Loss')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-10-23T19:50:02.079408Z","iopub.execute_input":"2023-10-23T19:50:02.079686Z","iopub.status.idle":"2023-10-23T19:50:02.777075Z","shell.execute_reply.started":"2023-10-23T19:50:02.079633Z","shell.execute_reply":"2023-10-23T19:50:02.775971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}