{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install facenet-pytorch","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:52:16.271916Z","iopub.execute_input":"2021-05-24T07:52:16.272224Z","iopub.status.idle":"2021-05-24T07:52:24.661650Z","shell.execute_reply.started":"2021-05-24T07:52:16.272174Z","shell.execute_reply":"2021-05-24T07:52:24.660602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from facenet_pytorch.models.inception_resnet_v1 import get_torch_home\ntorch_home = get_torch_home()","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:52:28.085439Z","iopub.execute_input":"2021-05-24T07:52:28.085781Z","iopub.status.idle":"2021-05-24T07:52:29.865998Z","shell.execute_reply.started":"2021-05-24T07:52:28.085723Z","shell.execute_reply":"2021-05-24T07:52:29.865249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport glob\nimport time\nimport torch\nimport cv2\nfrom PIL import Image\nimport numpy as np\nimport pandas as pd\nfrom matplotlib import pyplot as plt\nfrom tqdm.notebook import tqdm\n\n\nfrom facenet_pytorch import MTCNN, InceptionResnetV1, extract_face\n\ndevice = 'cuda:0' if torch.cuda.is_available() else 'cpu'\nprint(f'Running on device: {device}')","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:52:47.445804Z","iopub.execute_input":"2021-05-24T07:52:47.446105Z","iopub.status.idle":"2021-05-24T07:52:47.508333Z","shell.execute_reply.started":"2021-05-24T07:52:47.446054Z","shell.execute_reply":"2021-05-24T07:52:47.507261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mtcnn = MTCNN(margin=14, keep_all=True, factor=0.5, device=device).eval()\n\n\nresnet = InceptionResnetV1(pretrained='vggface2', device=device).eval()","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:53:06.500707Z","iopub.execute_input":"2021-05-24T07:53:06.501057Z","iopub.status.idle":"2021-05-24T07:53:13.375023Z","shell.execute_reply.started":"2021-05-24T07:53:06.500998Z","shell.execute_reply":"2021-05-24T07:53:13.374306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DetectionPipeline:\n\n    def __init__(self, detector, n_frames=None, batch_size=60, resize=None):\n\n        self.detector = detector\n        self.n_frames = n_frames\n        self.batch_size = batch_size\n        self.resize = resize\n        \n            \n    def __call__(self, filename):\n \n   \n        v_cap = cv2.VideoCapture(filename)\n        v_len = int(v_cap.get(cv2.CAP_PROP_FRAME_COUNT))\n        print(v_len)\n    \n        if self.n_frames is None:\n            sample = np.arange(0, v_len)\n        else:\n            sample = np.linspace(0, v_len - 1, self.n_frames).astype(int)\n\n   \n        faces = []\n        frames = []\n        for j in range(v_len):\n            success = v_cap.grab()\n            if j in sample:\n    \n                success, frame = v_cap.retrieve()\n                if not success:\n                    continue\n                frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n                frame = Image.fromarray(frame)\n                \n    \n                if self.resize is not None:\n                    frame = frame.resize([int(d * self.resize) for d in frame.size])\n                frames.append(frame)\n\n   \n                if len(frames) % self.batch_size == 0 or j == sample[-1]:\n                    faces.extend(self.detector(frames))\n                    frames = []\n\n        v_cap.release()\n        \n        return faces","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:54:10.025160Z","iopub.execute_input":"2021-05-24T07:54:10.025473Z","iopub.status.idle":"2021-05-24T07:54:10.037486Z","shell.execute_reply.started":"2021-05-24T07:54:10.025409Z","shell.execute_reply":"2021-05-24T07:54:10.036719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import copy\ndef process_faces(faces, resnet):\n\n    faces = [f for f in faces if f is not None]\n    if(len(faces) == 0):\n        return []\n\n    faces = torch.cat(faces).to(device)\n    if(len(faces)<290):\n        return []\n\n    faces = faces[:290]\n\n    embeddings = resnet(faces)\n\n    centroid = embeddings.mean(dim=0)\n    \n    x = (embeddings - centroid).norm(dim=1).cpu().numpy()\n    \n    return x\n","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:54:56.280587Z","iopub.execute_input":"2021-05-24T07:54:56.280881Z","iopub.status.idle":"2021-05-24T07:54:56.290235Z","shell.execute_reply.started":"2021-05-24T07:54:56.280835Z","shell.execute_reply":"2021-05-24T07:54:56.289573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"detection_pipeline = DetectionPipeline(detector=mtcnn, batch_size=60, resize=0.25)\nimport json\n\nwith open('../input/deepfake-detection-challenge/train_sample_videos/metadata.json') as f:\n  data = json.load(f)\n\nfilenames = glob.glob('/kaggle/input/deepfake-detection-challenge/train_sample_videos/*.mp4')\ntotal_files = len(filenames)\n\nX = []\ny = []\nstart = time.time()\nn_processed = 0\n\nwith torch.no_grad():\n    for i, filename in tqdm(enumerate(filenames), total=len(filenames)):\n        print(i, filename)\n        try:\n\n            faces = detection_pipeline(filename)\n                    \n            z = process_faces(faces, resnet)\n            if (len(z)!=0):\n                X.append(z)\n                if(data[filename[63:]]['label']=='FAKE'):\n                    y.append(1)\n                else:\n                    y.append(0)\n\n        except KeyboardInterrupt:\n            print('\\nStopped.')\n            break\n\n        except Exception as e:\n            print(e)\n            X.append(None)\n        \n        n_processed += len(faces)\n","metadata":{"execution":{"iopub.status.busy":"2021-05-24T07:55:54.326664Z","iopub.execute_input":"2021-05-24T07:55:54.326998Z","iopub.status.idle":"2021-05-24T08:35:07.802221Z","shell.execute_reply.started":"2021-05-24T07:55:54.326933Z","shell.execute_reply":"2021-05-24T08:35:07.801421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X=X[1:]\ny=y[1:]","metadata":{"execution":{"iopub.status.busy":"2021-05-24T08:35:24.410152Z","iopub.execute_input":"2021-05-24T08:35:24.410597Z","iopub.status.idle":"2021-05-24T08:35:24.414541Z","shell.execute_reply.started":"2021-05-24T08:35:24.410396Z","shell.execute_reply":"2021-05-24T08:35:24.413749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.85, random_state=5)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T08:35:26.751149Z","iopub.execute_input":"2021-05-24T08:35:26.751445Z","iopub.status.idle":"2021-05-24T08:35:28.032294Z","shell.execute_reply.started":"2021-05-24T08:35:26.751393Z","shell.execute_reply":"2021-05-24T08:35:28.031558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\n\nclf = LogisticRegression(random_state=0).fit(X_train, y_train)\ny_pred_lr = clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T08:35:31.704559Z","iopub.execute_input":"2021-05-24T08:35:31.704922Z","iopub.status.idle":"2021-05-24T08:35:31.827773Z","shell.execute_reply.started":"2021-05-24T08:35:31.704866Z","shell.execute_reply":"2021-05-24T08:35:31.826099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nprint(accuracy_score(y_test,y_pred_lr))","metadata":{"execution":{"iopub.status.busy":"2021-05-24T08:35:35.808917Z","iopub.execute_input":"2021-05-24T08:35:35.809222Z","iopub.status.idle":"2021-05-24T08:35:35.816907Z","shell.execute_reply.started":"2021-05-24T08:35:35.809170Z","shell.execute_reply":"2021-05-24T08:35:35.816080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(y_test)):\n        if y_test[i] == y_pred_lr[i] and y_test[i]==0:\n            print(y_test[i])","metadata":{"execution":{"iopub.status.busy":"2021-05-24T08:35:38.919061Z","iopub.execute_input":"2021-05-24T08:35:38.919351Z","iopub.status.idle":"2021-05-24T08:35:38.925662Z","shell.execute_reply.started":"2021-05-24T08:35:38.919301Z","shell.execute_reply":"2021-05-24T08:35:38.924818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\n\ngnb = GaussianNB()\ny_pred_gnb = gnb.fit(X_train, y_train).predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T08:35:42.813931Z","iopub.execute_input":"2021-05-24T08:35:42.814250Z","iopub.status.idle":"2021-05-24T08:35:42.826846Z","shell.execute_reply.started":"2021-05-24T08:35:42.814199Z","shell.execute_reply":"2021-05-24T08:35:42.825803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nprint(accuracy_score(y_test,y_pred_gnb))","metadata":{"execution":{"iopub.status.busy":"2021-05-24T08:35:46.609132Z","iopub.execute_input":"2021-05-24T08:35:46.609416Z","iopub.status.idle":"2021-05-24T08:35:46.618037Z","shell.execute_reply.started":"2021-05-24T08:35:46.609363Z","shell.execute_reply":"2021-05-24T08:35:46.614815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import svm\n\nclf = svm.SVC(gamma='auto')\nclf.fit(X_train, y_train)\ny_pred_svm = clf.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T08:35:50.254109Z","iopub.execute_input":"2021-05-24T08:35:50.254397Z","iopub.status.idle":"2021-05-24T08:35:50.266096Z","shell.execute_reply.started":"2021-05-24T08:35:50.254347Z","shell.execute_reply":"2021-05-24T08:35:50.265221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\nprint(accuracy_score(y_test,y_pred_svm))","metadata":{"execution":{"iopub.status.busy":"2021-05-24T08:35:53.829218Z","iopub.execute_input":"2021-05-24T08:35:53.829524Z","iopub.status.idle":"2021-05-24T08:35:53.834844Z","shell.execute_reply.started":"2021-05-24T08:35:53.829452Z","shell.execute_reply":"2021-05-24T08:35:53.833878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}