{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# DeepFake detection","metadata":{}},{"cell_type":"markdown","source":"### Installation des packages","metadata":{}},{"cell_type":"code","source":"pip install opencv-python","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:22:48.751667Z","iopub.execute_input":"2022-03-04T14:22:48.752106Z","iopub.status.idle":"2022-03-04T14:22:57.073458Z","shell.execute_reply.started":"2022-03-04T14:22:48.752026Z","shell.execute_reply":"2022-03-04T14:22:57.072525Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install tensorflow","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:22:57.075898Z","iopub.execute_input":"2022-03-04T14:22:57.076317Z","iopub.status.idle":"2022-03-04T14:23:03.278559Z","shell.execute_reply.started":"2022-03-04T14:22:57.076244Z","shell.execute_reply":"2022-03-04T14:23:03.277257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install mtcnn","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:03.280503Z","iopub.execute_input":"2022-03-04T14:23:03.280912Z","iopub.status.idle":"2022-03-04T14:23:11.899718Z","shell.execute_reply.started":"2022-03-04T14:23:03.280835Z","shell.execute_reply":"2022-03-04T14:23:11.89821Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Import","metadata":{}},{"cell_type":"code","source":"# data\nimport numpy as np\nimport pandas as pd\n\n# visulazation\nimport matplotlib.pyplot as plt\n\n# image processing\nimport cv2\n\n# extracting zippped file\nimport tarfile\n\n# shuffle df  \nfrom sklearn.utils import shuffle\n#df = shuffle(df)\n\n# directory\nimport os\nimport shutil\n\n# croping?\nfrom PIL import Image \n\n# face detection with MTCNN\nfrom mtcnn import MTCNN\n\nimport tensorflow as tf\n\n# neural network\nfrom tensorflow.keras.layers import Input, Dense, Flatten, Conv2D, MaxPooling2D, BatchNormalization, Dropout, Reshape, Concatenate, LeakyReLU\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:11.901994Z","iopub.execute_input":"2022-03-04T14:23:11.902333Z","iopub.status.idle":"2022-03-04T14:23:18.715075Z","shell.execute_reply.started":"2022-03-04T14:23:11.902275Z","shell.execute_reply":"2022-03-04T14:23:18.714088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# hide warnings tensorflow\n#tf.logging.set_verbosity(tf.logging.ERROR)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:18.719035Z","iopub.execute_input":"2022-03-04T14:23:18.719395Z","iopub.status.idle":"2022-03-04T14:23:18.723438Z","shell.execute_reply.started":"2022-03-04T14:23:18.719319Z","shell.execute_reply":"2022-03-04T14:23:18.722379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Avancement","metadata":{}},{"cell_type":"markdown","source":"### A faire \n- ~~supprimer les cadres autour des visages (vérifier la taille des images croppées)~~\n- ~~aggréger les résultats images > video~~\n- ~~générer l'output attendu par Kaggle (filename, label (float))~~\n- ajouter des données depuis les différentes datasets splitted, afin d'avoir au moins 50% de reals\n- resize / normalize les images croppées\n- model EfficientNet (pre-trained)\n- transfer learning (rajouter des couches à la fin du réseau de neuronnes pour améliorer le modèle\n- matrice de confusion pour les résultats","metadata":{}},{"cell_type":"markdown","source":"### Traitement des vidéos","metadata":{}},{"cell_type":"code","source":"# jeux de données d'entrainement et de test\ntrain_videos_path = r'D:\\videos\\train_sample_videos\\\\'\ntest_videos_path = r'D:\\videos\\test_sample_videos\\\\'\n\n# dossiers remplis à partir du jeu d'entrainement et des metadatas\nreal_videos_path = r'D:\\videos\\real_videos\\\\'\nfake_videos_path = r'D:\\videos\\fake_videos\\\\'\n\n# dossiers remplis à partir du découpage en frames des videos contenues dans les dossiers ci-dessus\nreal_frames_path = r'D:\\videos\\frames\\real_frames\\\\'\nfake_frames_path = r'D:\\videos\\frames\\fake_frames\\\\'\n\n# frames cropped\nreal_frames_cropped_path = r'D:\\videos\\frames_cropped\\real_frames\\\\'\nfake_frames_cropped_path = r'D:\\videos\\frames_cropped\\fake_frames\\\\'","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:18.727534Z","iopub.execute_input":"2022-03-04T14:23:18.727931Z","iopub.status.idle":"2022-03-04T14:23:18.737692Z","shell.execute_reply.started":"2022-03-04T14:23:18.727858Z","shell.execute_reply":"2022-03-04T14:23:18.73696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fonction qui permet d'associer les videos aux labels\ndef get_meta_from_json(path):\n    df = pd.read_json(path)\n    df = df.T\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:18.739855Z","iopub.execute_input":"2022-03-04T14:23:18.740627Z","iopub.status.idle":"2022-03-04T14:23:18.752631Z","shell.execute_reply.started":"2022-03-04T14:23:18.74056Z","shell.execute_reply":"2022-03-04T14:23:18.751572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# on associe les labels aux videos d'entrainement\nmeta_train_df = get_meta_from_json(train_videos_path+'metadata.json')\nmeta_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:18.756297Z","iopub.execute_input":"2022-03-04T14:23:18.757172Z","iopub.status.idle":"2022-03-04T14:23:19.39116Z","shell.execute_reply.started":"2022-03-04T14:23:18.757102Z","shell.execute_reply":"2022-03-04T14:23:19.38854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# visualisation de la répartition des données\nmeta_train_df.groupby('label')['label'].count().plot(figsize=(15, 5), kind='bar', title='Répartition des vidéos en fonction des labels')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.392192Z","iopub.status.idle":"2022-03-04T14:23:19.393055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# on place les videos dans les dossiers correspondants\nfor index, row in meta_train_df.iterrows():\n    if row.label == 'FAKE':\n        shutil.copyfile(train_videos_path+index, fake_videos_path+index)\n    elif row.label == 'REAL':\n        shutil.copyfile(train_videos_path+index, real_videos_path+index)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.394565Z","iopub.status.idle":"2022-03-04T14:23:19.395131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# récupérer une frame sur une video\ndef getFrame(sec, video_name, video, frame_folder, count):\n    video.set(cv2.CAP_PROP_POS_MSEC,sec*1000)\n    hasFrames,image = video.read()  \n    str_video_name = str(video_name)[0:-4]\n    if hasFrames:\n        cv2.imwrite(frame_folder + str_video_name + \"_frame%d.jpg\" % count, image)\n    return hasFrames","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.396618Z","iopub.status.idle":"2022-03-04T14:23:19.397367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# decoupage des videos (framerate réglable)\ndef videosToFrames(videos_path, frames_path): \n    for filename in os.listdir(videos_path):\n        if filename.endswith(\".mp4\"):\n            sec = 0\n            frameRate = 1 # 1 image par X seconde(s)\n            count = 1\n            video = cv2.VideoCapture(videos_path + filename)\n            success = getFrame(sec, str(filename), video, frames_path, count)\n            while success:\n                count = count + 1\n                sec = sec + frameRate\n                sec = round(sec, 2)\n                success = getFrame(sec, str(filename), video, frames_path, count)\n        else:\n            continue","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.398984Z","iopub.status.idle":"2022-03-04T14:23:19.399469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# on découpe les videos fake et real puis on place les images dans les dossiers correspondants\nvideosToFrames(fake_videos_path, fake_frames_path)\nvideosToFrames(real_videos_path, real_frames_path)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.400923Z","iopub.status.idle":"2022-03-04T14:23:19.401308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def nbTotalFrames(frames_paths):\n    for frames_path in frames_paths:\n        path, dirs, files = next(os.walk(frames_path))\n        file_count = len(files)\n        print(\"Fakes: {}\".format(file_count)) if frames_path == fake_frames_path else print(\"Reals: {}\".format(file_count))","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.402314Z","iopub.status.idle":"2022-03-04T14:23:19.402809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nbTotalFrames([fake_frames_path, real_frames_path])","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.404Z","iopub.status.idle":"2022-03-04T14:23:19.404536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualisation des frames","metadata":{}},{"cell_type":"code","source":"# visualisation des images\ndef show_image(image):\n    plt.figure(figsize=(18,15))\n    #Before showing image, bgr color order transformed to rgb order\n    plt.imshow(cv2.cvtColor(image, cv2.COLOR_BGR2RGB))\n    plt.xticks([])\n    plt.yticks([])\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.405704Z","iopub.status.idle":"2022-03-04T14:23:19.406238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_image = cv2.imread(real_frames_path+\"bdnaqemxmr_frame5.jpg\")\nshow_image(real_image)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.407266Z","iopub.status.idle":"2022-03-04T14:23:19.407746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_image = cv2.imread(fake_frames_path+\"aagfhgtpmv_frame5.jpg\")\nshow_image(fake_image)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.408774Z","iopub.status.idle":"2022-03-04T14:23:19.409225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_crop_image = real_image[x:w, y:h]\n#show_image(test_crop_image)\ndef center_crop(img):\n    height, width = img.shape[0], img.shape[1]\n    if(width > height):\n        crop_width = crop_height = height        \n    else:\n        crop_height = crop_width = width\n    mid_x, mid_y = int(width/2), int(height/2)\n    cw2, ch2 = int(crop_width/2), int(crop_height/2)\n    crop_img = img[mid_y-ch2:mid_y+ch2, mid_x-cw2:mid_x+cw2]\n    #cv2.imshow(\"cropped\", crop_img)\n    return crop_img","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.41038Z","iopub.status.idle":"2022-03-04T14:23:19.410972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Detection des visages","metadata":{}},{"cell_type":"code","source":"def face_detector(image):\n    detector = MTCNN()\n    detection = detector.detect_faces(image)\n    \n    if(len(detection) > 0):    \n        bounding_box = detection[0]['box']\n        keypoints = detection[0]['keypoints']\n\n        cv2.rectangle(image,\n                  (bounding_box[0], bounding_box[1]), # point de départ\n                  (bounding_box[0]+bounding_box[2], bounding_box[1] + bounding_box[3]), # point d'arrivée\n                  (0,155,255), # color\n                  1) # thickness\n        cv2.circle(image,(keypoints['left_eye']), 2, (255,0,0), 2)\n        cv2.circle(image,(keypoints['right_eye']), 2, (255,0,0), 2)\n        cv2.circle(image,(keypoints['nose']), 2, (0,0,255), 2)\n        cv2.circle(image,(keypoints['mouth_left']), 2, (0,255,0), 2)\n        cv2.circle(image,(keypoints['mouth_right']), 2, (0,255,0), 2)\n\n        show_image(image)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.411823Z","iopub.status.idle":"2022-03-04T14:23:19.412172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# permet de cacher les warnings de tensorflow\nimport logging\nlogger = tf.get_logger()\nlogger.setLevel(logging.ERROR)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.413122Z","iopub.status.idle":"2022-03-04T14:23:19.413681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"real_image_test = cv2.imread(real_frames_path+\"bdnaqemxmr_frame5.jpg\")\nface_detector(real_image_test)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.41476Z","iopub.status.idle":"2022-03-04T14:23:19.415207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fake_image_test = cv2.imread(fake_frames_path+\"aagfhgtpmv_frame5.jpg\")\nface_detector(fake_image_test)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.416203Z","iopub.status.idle":"2022-03-04T14:23:19.416816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Croper les frames","metadata":{}},{"cell_type":"code","source":"def visu_crop_face(image):\n    detector = MTCNN()\n    detection = detector.detect_faces(image)\n    \n    if(len(detection) > 0):\n        bounding_box = detection[0]['box']\n        keypoints = detection[0]['keypoints']\n\n        x = bounding_box[0]\n        y = bounding_box[1]\n        h = bounding_box[2]\n        w = bounding_box[3]\n\n        #print(bounding_box)\n        #print(keypoints)\n\n        image_copy = image.copy() \n        crop_image = image_copy[y-int(h*0.2):y+w+int(h*0.2), x-int(w*0.2):x+h+int(w*0.2)]\n        #square = np.ceil(np.sqrt(crop_image.size[0]*crop_image.size[1])).astype(int)\n        #square_image = crop_image.resize((square, square))\n        #crop_img = image_copy[y-h:y+w+h, x-w:x+h+w]\n        return crop_image\n        # show_image(crop_image)\n\n        #image_copy = image.copy()\n\n        #for (x, y, w, h) in face:\n         #   crop_img = image_copy[y:y+h, x:x+w]\n          #  break;\n\n        #show_image(crop_img)\n            # cv2.rectangle(image_copy.(x, y), (x+w, y+h), (255, 0, 0), 3)\n    else:\n        return 'no face detected'\n    ","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.417973Z","iopub.status.idle":"2022-03-04T14:23:19.418588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_image(visu_crop_face(real_image))","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.419704Z","iopub.status.idle":"2022-03-04T14:23:19.420216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_image(visu_crop_face(fake_image))","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.421381Z","iopub.status.idle":"2022-03-04T14:23:19.421843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fake_crop_size = visu_crop_face(fake_image)\n#width, height = fake_crop_size.size\n#print(width, height)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.422791Z","iopub.status.idle":"2022-03-04T14:23:19.423308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":" ### On place les faces dans des dossiers","metadata":{}},{"cell_type":"code","source":"def framesToFaces(frames_path, frames_cropped_path): \n    for filename in os.listdir(frames_path):\n        image = cv2.imread(frames_path + filename)\n        cropped_img = visu_crop_face(image)\n        if cropped_img != 'no face detected':\n            cv2.imwrite(frames_cropped_path + filename, cropped_img)\n            print(filename + 'OK')\n        else:\n            print(filename + 'KO')\n            #continue","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.424305Z","iopub.status.idle":"2022-03-04T14:23:19.424956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"framesToFaces(fake_frames_path, fake_frames_cropped_path)\n#framesToFaces(real_frames_path, real_frames_cropped_path)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.426313Z","iopub.status.idle":"2022-03-04T14:23:19.426825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model","metadata":{}},{"cell_type":"code","source":"# Create a Classifier class\nclass Classifier:\n    def __init__():\n        self.model = 0\n    \n    def predict(self, x):\n        return self.model.predict(x)\n    \n    def fit(self, x, y):\n        return self.model.train_on_batch(x, y)\n    \n    def get_accuracy(self, x, y):\n        return self.model.test_on_batch(x, y)\n    \n    def load(self, path):\n        self.model.load_weights(path)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.427771Z","iopub.status.idle":"2022-03-04T14:23:19.42831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 256x256 RGB\nimage_dimensions = {'height':256, 'width':256, 'channels':3}","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.42937Z","iopub.status.idle":"2022-03-04T14:23:19.42993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a MesoNet class using the Classifier\n\nclass Meso4(Classifier):\n    def __init__(self, learning_rate = 0.001):\n        self.model = self.init_model()\n        optimizer = Adam(lr = learning_rate)\n        self.model.compile(optimizer = optimizer,\n                           loss = 'mean_squared_error', #binary_crossentropy\n                           metrics = ['accuracy'])\n    \n    def init_model(self): \n        x = Input(shape = (image_dimensions['height'],\n                           image_dimensions['width'],\n                           image_dimensions['channels']))\n        \n        x1 = Conv2D(8, (3, 3), padding='same', activation = 'relu')(x)\n        x1 = BatchNormalization()(x1)\n        x1 = MaxPooling2D(pool_size=(2, 2), padding='same')(x1)\n        \n        x2 = Conv2D(8, (5, 5), padding='same', activation = 'relu')(x1)\n        x2 = BatchNormalization()(x2)\n        x2 = MaxPooling2D(pool_size=(2, 2), padding='same')(x2)\n        \n        x3 = Conv2D(16, (5, 5), padding='same', activation = 'relu')(x2)\n        x3 = BatchNormalization()(x3)\n        x3 = MaxPooling2D(pool_size=(2, 2), padding='same')(x3)\n        \n        x4 = Conv2D(16, (5, 5), padding='same', activation = 'relu')(x3)\n        x4 = BatchNormalization()(x4)\n        x4 = MaxPooling2D(pool_size=(4, 4), padding='same')(x4)\n        \n        y = Flatten()(x4)\n        y = Dropout(0.5)(y)\n        y = Dense(16)(y)\n        y = LeakyReLU(alpha=0.1)(y)\n        y = Dropout(0.5)(y)\n        y = Dense(1, activation = 'sigmoid')(y) #softmax\n\n        return Model(inputs = x, outputs = y)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.43093Z","iopub.status.idle":"2022-03-04T14:23:19.431407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### MesoNet","metadata":{}},{"cell_type":"code","source":"# chargement du Model MesoNet avec des poids\nmeso = Meso4()\nmeso.load('weight/Meso4_DF.h5')","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.432276Z","iopub.status.idle":"2022-03-04T14:23:19.432877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# On prépare l'image\n# 1920x1080\ndataGenerator = ImageDataGenerator(rescale=1./255)\n# dossier data -> sous-dossier DeepFake / Real\ngenerator = dataGenerator.flow_from_directory(\n    r'D:\\videos\\frames_cropped\\\\',\n    target_size=(256, 256),\n    batch_size=1,\n    class_mode='binary')","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.433919Z","iopub.status.idle":"2022-03-04T14:23:19.434481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# nos classes\ngenerator.class_indices","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.435428Z","iopub.status.idle":"2022-03-04T14:23:19.436049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# '.ipynb_checkpoints' is a *hidden* file Jupyter creates for autosaves\n# It must be removed for flow_from_directory to work.\n# Equivalent command in Unix (for Mac / Linux users)\n# !rm -r /Users/USER_NAME/mesonet/mesonet/data/.ipynb_checkpoints/\n\n# !rmdir /s /q r'C:\\Users\\Yanis\\Desktop\\Data Science\\Ydays\\data\\.ipynb_checkpoints'","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.436977Z","iopub.status.idle":"2022-03-04T14:23:19.437504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Rendering dans MesoNet\n# X = image y = label\nX, y = generator.next()\n\n# Evaluation des predictions\nprint(f\"Taux de confiance image sans deepfake: {meso.predict(X)[0][0]:.4f}\")\nprint(f\"\\nVrai label: {int(y[0])}\")\nprint(f\"\\nPrediction correcte: {round(meso.predict(X)[0][0])==y[0]}\")\n\n# predict proba\n\n# plt.imshow(np.squeeze(X));\n\n# print(X[0])","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.438423Z","iopub.status.idle":"2022-03-04T14:23:19.43882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# On classe les images par résultat de la classification\ncorrect_real = []\ncorrect_real_pred = []\n\ncorrect_deepfake = []\ncorrect_deepfake_pred = []\n\nmisclassified_real = []\nmisclassified_real_pred = []\n\nmisclassified_deepfake = []\nmisclassified_deepfake_pred = []\n\ntest_X = []\nall_preds = []","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.440023Z","iopub.status.idle":"2022-03-04T14:23:19.440539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generating predictions on validation set, storing in separate lists\nfor i in range(len(generator.labels)):\n    \n    # Loading next picture, generating prediction\n    X, y = generator.next()\n    pred = meso.predict(X)[0][0]\n    \n    # Sorting into proper category\n    if round(pred)==y[0] and y[0]==1:\n        correct_real.append(X)\n        correct_real_pred.append(pred)\n        all_preds.append(pred)\n    elif round(pred)==y[0] and y[0]==0:\n        correct_deepfake.append(X)\n        correct_deepfake_pred.append(pred)\n        all_preds.append(pred)\n    elif y[0]==1:\n        misclassified_real.append(X)\n        misclassified_real_pred.append(pred)\n        all_preds.append(pred)\n    else:\n        misclassified_deepfake.append(X)\n        misclassified_deepfake_pred.append(pred)\n        all_preds.append(pred)\n        \n    # Printing status update\n    if i % 250 == 0:\n        print(i, ' predictions ...')\n    \n    if i == len(generator.labels)-1:\n        print(\"les\", len(generator.labels), \"predictions sont terminées\")","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.441704Z","iopub.status.idle":"2022-03-04T14:23:19.442155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(all_preds)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.443144Z","iopub.status.idle":"2022-03-04T14:23:19.443617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(all_preds)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.44476Z","iopub.status.idle":"2022-03-04T14:23:19.445145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualisation","metadata":{}},{"cell_type":"code","source":"def plotter(images,preds):\n    fig = plt.figure(figsize=(16,9))\n    subset = np.random.randint(0, len(images)-1, 12)\n    for i,j in enumerate(subset):\n        fig.add_subplot(3,4,i+1)\n        plt.imshow(np.squeeze(images[j]))\n        plt.xlabel(f\"Model confidence: \\n{preds[j]:.4f}\")\n        plt.tight_layout()\n        ax = plt.gca()\n        ax.axes.xaxis.set_ticks([])\n        ax.axes.yaxis.set_ticks([])\n    plt.show;\n    return","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.446078Z","iopub.status.idle":"2022-03-04T14:23:19.446599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### REAL : prediction correcte","metadata":{}},{"cell_type":"code","source":"plotter(correct_real, correct_real_pred)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.44751Z","iopub.status.idle":"2022-03-04T14:23:19.448004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nb_real_preds = len(correct_real_pred)\n\nlist_real_frames = os.listdir(real_frames_cropped_path)\nnb_real_frames = len(list_real_frames)\n\nprint(\"pourcentage real pred : \" + str(round(nb_real_preds/nb_real_frames*100, 2)) + \"%\")\nprint(\"real pred : \" + str(nb_real_preds))\nprint(\"real total : \" + str(nb_real_frames))","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.448975Z","iopub.status.idle":"2022-03-04T14:23:19.449501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### REAL : mauvaise prediction","metadata":{}},{"cell_type":"code","source":"plotter(misclassified_real, misclassified_real_pred)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.450421Z","iopub.status.idle":"2022-03-04T14:23:19.450987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nb_misclassified_real_preds = len(misclassified_real_pred)\nprint(\"pourcentage bad real pred : \" + str(round(nb_misclassified_real_preds/nb_real_frames*100, 2)) + \"%\")\nprint(\"bad real pred : \" + str(nb_misclassified_real_preds))\nprint(\"real total : \" + str(nb_real_frames))","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.452154Z","iopub.status.idle":"2022-03-04T14:23:19.452808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### DEEPFAKE : prediction correcte","metadata":{}},{"cell_type":"code","source":"plotter(correct_deepfake, correct_deepfake_pred)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.45405Z","iopub.status.idle":"2022-03-04T14:23:19.454562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nb_fake_preds = len(correct_deepfake_pred)\n\nlist_fake_frames = os.listdir(fake_frames_cropped_path)\nnb_fake_frames = len(list_fake_frames)\n\nprint(\"pourcentage fake pred : \" + str(round(nb_fake_preds/nb_fake_frames*100, 2)) + \"%\")\nprint(\"real fake pred : \" + str(nb_fake_preds))\nprint(\"fake total : \" + str(nb_fake_frames))","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.45553Z","iopub.status.idle":"2022-03-04T14:23:19.456287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### DEEPFAKE : mauvaise prediction","metadata":{}},{"cell_type":"code","source":"plotter(misclassified_deepfake, misclassified_deepfake_pred)","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.457611Z","iopub.status.idle":"2022-03-04T14:23:19.458185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nb_misclassified_fake_preds = len(misclassified_deepfake_pred)\nprint(\"pourcentage bad fake pred : \" + str(round(nb_misclassified_fake_preds/nb_fake_frames*100, 2)) + \"%\")\nprint(\"bad fake pred : \" + str(nb_misclassified_fake_preds))\nprint(\"fake total : \" + str(nb_fake_frames))","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.459082Z","iopub.status.idle":"2022-03-04T14:23:19.459792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"correct_preds = nb_real_preds + nb_fake_preds\ntotal = nb_real_frames + nb_fake_frames\n\nprint(\"resultat (frames) \" + str(round(correct_preds/total*100, 2)) + \"%\")","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.460934Z","iopub.status.idle":"2022-03-04T14:23:19.461453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\nfrom sklearn import svm, datasets\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import plot_confusion_matrix\n\n# import some data to play with\niris = datasets.load_iris()\nX = iris.data\ny = iris.target\nclass_names = iris.target_names\n\n# Split the data into a training set and a test set\nX_train, X_test, y_train, y_test = train_test_split(X, y, random_state=0)\n\n# Run classifier, using a model that is too regularized (C too low) to see\n# the impact on the results\nclassifier = svm.SVC(kernel='linear', C=0.01).fit(X_train, y_train)\n\nnp.set_printoptions(precision=2)\n\n# Plot non-normalized confusion matrix\ntitles_options = [(\"Confusion matrix, without normalization\", None),\n                  (\"Normalized confusion matrix\", 'true')]\nfor title, normalize in titles_options:\n    disp = plot_confusion_matrix(classifier, X_test, y_test,\n                                 display_labels=class_names,\n                                 cmap=plt.cm.Blues,\n                                 normalize=normalize)\n    disp.ax_.set_title(title)\n\n    print(title)\n    print(disp.confusion_matrix)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.462432Z","iopub.status.idle":"2022-03-04T14:23:19.462871Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Resultat par video ","metadata":{}},{"cell_type":"code","source":"from os import listdir\nfrom os.path import isfile, join\n\nfake_files = [f for f in listdir(fake_frames_cropped_path) if isfile(join(fake_frames_cropped_path, f))]\nreal_files = [f for f in listdir(real_frames_cropped_path) if isfile(join(real_frames_cropped_path, f))]\n\nfiles = fake_files + real_files\ntest_df = pd.DataFrame({'filename': files, 'label': all_preds})\n\ntest_df['filename'] = test_df['filename'].map(lambda x : x.rstrip('.jpg').rstrip('1234567890').rstrip('1234567890').rstrip('_frame') + '.mp4')\n\nfinal_df = test_df.groupby('filename', as_index='filename').mean()\nfinal_df\n#final_df.index.name = None\n#final_df.size","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.463793Z","iopub.status.idle":"2022-03-04T14:23:19.464239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#meta_train_df.Index.duplicated(keep=first)\n#.size\n\nmeta_train_df.duplicated(i_eta_train_df.size)\n\n#rowData = meta_train_df.loc[['eudeqjhdfd.mp4'], :]\n#rowData","metadata":{"execution":{"iopub.status.busy":"2022-03-04T14:23:19.465118Z","iopub.status.idle":"2022-03-04T14:23:19.465749Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]}]}