{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":888125,"sourceType":"datasetVersion","datasetId":458848},{"sourceId":904207,"sourceType":"datasetVersion","datasetId":484608},{"sourceId":914842,"sourceType":"datasetVersion","datasetId":491770},{"sourceId":955742,"sourceType":"datasetVersion","datasetId":513681}],"dockerImageVersionId":29845,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Inference for Xception baseline model\n### Forked from: https://www.kaggle.com/humananalog/inference-demo\n### View this kernal for training of this model: https://www.kaggle.com/greatgamedota/xception-binary-classifier-with-ffhq-training\n### This kernal takes ~3-3.5 hours to submit with GPU","metadata":{}},{"cell_type":"code","source":"!pip install ../input/pytorchcv/pytorchcv-0.0.55-py2.py3-none-any.whl --quiet","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:20.299025Z","iopub.execute_input":"2024-08-31T16:19:20.299331Z","iopub.status.idle":"2024-08-31T16:19:48.492914Z","shell.execute_reply.started":"2024-08-31T16:19:20.299283Z","shell.execute_reply":"2024-08-31T16:19:48.491910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, sys, time\nimport cv2\nimport numpy as np\nimport pandas as pd\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\n%matplotlib inline\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-08-31T16:19:48.495351Z","iopub.execute_input":"2024-08-31T16:19:48.495639Z","iopub.status.idle":"2024-08-31T16:19:50.403423Z","shell.execute_reply.started":"2024-08-31T16:19:48.495596Z","shell.execute_reply":"2024-08-31T16:19:50.402543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dir = \"/kaggle/input/deepfake-detection-challenge/test_videos/\"\n\ntest_videos = sorted([x for x in os.listdir(test_dir) if x[-4:] == \".mp4\"])\nlen(test_videos)","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2024-08-31T16:19:50.404775Z","iopub.execute_input":"2024-08-31T16:19:50.405006Z","iopub.status.idle":"2024-08-31T16:19:50.517940Z","shell.execute_reply.started":"2024-08-31T16:19:50.404967Z","shell.execute_reply":"2024-08-31T16:19:50.517104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gpu = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:50.519601Z","iopub.execute_input":"2024-08-31T16:19:50.519932Z","iopub.status.idle":"2024-08-31T16:19:50.565830Z","shell.execute_reply.started":"2024-08-31T16:19:50.519871Z","shell.execute_reply":"2024-08-31T16:19:50.564756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import sys\nsys.path.insert(0, \"/kaggle/input/blazeface-pytorch\")\nsys.path.insert(0, \"/kaggle/input/deepfakes-inference-demo\")","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:50.568736Z","iopub.execute_input":"2024-08-31T16:19:50.568999Z","iopub.status.idle":"2024-08-31T16:19:50.575720Z","shell.execute_reply.started":"2024-08-31T16:19:50.568955Z","shell.execute_reply":"2024-08-31T16:19:50.574944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from blazeface import BlazeFace\nfacedet = BlazeFace().to(gpu)\nfacedet.load_weights(\"/kaggle/input/blazeface-pytorch/blazeface.pth\")\nfacedet.load_anchors(\"/kaggle/input/blazeface-pytorch/anchors.npy\")\n_ = facedet.train(False)","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:50.578468Z","iopub.execute_input":"2024-08-31T16:19:50.578798Z","iopub.status.idle":"2024-08-31T16:19:55.077511Z","shell.execute_reply.started":"2024-08-31T16:19:50.578742Z","shell.execute_reply":"2024-08-31T16:19:55.076264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from helpers.read_video_1 import VideoReader\nfrom helpers.face_extract_1 import FaceExtractor\n\nframes_per_video = 20\n\nvideo_reader = VideoReader()\nvideo_read_fn = lambda x: video_reader.read_frames(x, num_frames=frames_per_video)\nface_extractor = FaceExtractor(video_read_fn, facedet)","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:55.079308Z","iopub.execute_input":"2024-08-31T16:19:55.079742Z","iopub.status.idle":"2024-08-31T16:19:55.107386Z","shell.execute_reply.started":"2024-08-31T16:19:55.079677Z","shell.execute_reply":"2024-08-31T16:19:55.106653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_size = 150","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:55.109022Z","iopub.execute_input":"2024-08-31T16:19:55.109404Z","iopub.status.idle":"2024-08-31T16:19:55.113888Z","shell.execute_reply.started":"2024-08-31T16:19:55.109332Z","shell.execute_reply":"2024-08-31T16:19:55.113000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torchvision.transforms import Normalize\n\nmean = [0.485, 0.456, 0.406]\nstd = [0.229, 0.224, 0.225]\nnormalize_transform = Normalize(mean, std)","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:55.115777Z","iopub.execute_input":"2024-08-31T16:19:55.116140Z","iopub.status.idle":"2024-08-31T16:19:55.251104Z","shell.execute_reply.started":"2024-08-31T16:19:55.116074Z","shell.execute_reply":"2024-08-31T16:19:55.250407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def isotropically_resize_image(img, size, resample=cv2.INTER_AREA):\n    h, w = img.shape[:2]\n    if w > h:\n        h = h * size // w\n        w = size\n    else:\n        w = w * size // h\n        h = size\n\n    resized = cv2.resize(img, (w, h), interpolation=resample)\n    return resized\n\n\ndef make_square_image(img):\n    h, w = img.shape[:2]\n    size = max(h, w)\n    t = 0\n    b = size - h\n    l = 0\n    r = size - w\n    return cv2.copyMakeBorder(img, t, b, l, r, cv2.BORDER_CONSTANT, value=0)","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:55.252755Z","iopub.execute_input":"2024-08-31T16:19:55.253107Z","iopub.status.idle":"2024-08-31T16:19:55.263093Z","shell.execute_reply.started":"2024-08-31T16:19:55.253040Z","shell.execute_reply":"2024-08-31T16:19:55.262222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pytorchcv.model_provider import get_model\nmodel = get_model(\"xception\", pretrained=False)\nmodel = nn.Sequential(*list(model.children())[:-1]) # Remove original output layer\n\nmodel[0].final_block.pool = nn.Sequential(nn.AdaptiveAvgPool2d(1))\n\nclass Head(torch.nn.Module):\n  def __init__(self, in_f, out_f):\n    super(Head, self).__init__()\n    \n    self.f = nn.Flatten()\n    self.l = nn.Linear(in_f, 512)\n    self.d = nn.Dropout(0.5)\n    self.o = nn.Linear(512, out_f)\n    self.b1 = nn.BatchNorm1d(in_f)\n    self.b2 = nn.BatchNorm1d(512)\n    self.r = nn.ReLU()\n\n  def forward(self, x):\n    x = self.f(x)\n    x = self.b1(x)\n    x = self.d(x)\n\n    x = self.l(x)\n    x = self.r(x)\n    x = self.b2(x)\n    x = self.d(x)\n\n    out = self.o(x)\n    return out\n\nclass FCN(torch.nn.Module):\n  def __init__(self, base, in_f):\n    super(FCN, self).__init__()\n    self.base = base\n    self.h1 = Head(in_f, 1)\n  \n  def forward(self, x):\n    x = self.base(x)\n    return self.h1(x)\n\nnet = []\nmodel = FCN(model, 2048)\nmodel = model.cuda()\nmodel.load_state_dict(torch.load('../input/deepfake-xception/modelv2.pth'))\nnet.append(model)","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:55.264524Z","iopub.execute_input":"2024-08-31T16:19:55.264801Z","iopub.status.idle":"2024-08-31T16:19:56.575789Z","shell.execute_reply.started":"2024-08-31T16:19:55.264754Z","shell.execute_reply":"2024-08-31T16:19:56.574853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction loop","metadata":{}},{"cell_type":"code","source":"def predict_on_video(video_path, batch_size):\n    try:\n        # Find the faces for N frames in the video.\n        faces = face_extractor.process_video(video_path)\n\n        # Only look at one face per frame.\n        face_extractor.keep_only_best_face(faces)\n        \n        if len(faces) > 0:\n            # NOTE: When running on the CPU, the batch size must be fixed\n            # or else memory usage will blow up. (Bug in PyTorch?)\n            x = np.zeros((batch_size, input_size, input_size, 3), dtype=np.uint8)\n\n            # If we found any faces, prepare them for the model.\n            n = 0\n            for frame_data in faces:\n                for face in frame_data[\"faces\"]:\n                    # Resize to the model's required input size.\n                    # We keep the aspect ratio intact and add zero\n                    # padding if necessary.                    \n                    resized_face = isotropically_resize_image(face, input_size)\n                    resized_face = make_square_image(resized_face)\n\n                    if n < batch_size:\n                        x[n] = resized_face\n                        n += 1\n                    else:\n                        print(\"WARNING: have %d faces but batch size is %d\" % (n, batch_size))\n                    \n                    # Test time augmentation: horizontal flips.\n                    # TODO: not sure yet if this helps or not\n                    #x[n] = cv2.flip(resized_face, 1)\n                    #n += 1\n\n            if n > 0:\n                x = torch.tensor(x, device=gpu).float()\n\n                # Preprocess the images.\n                x = x.permute((0, 3, 1, 2))\n\n                for i in range(len(x)):\n                    x[i] = normalize_transform(x[i] / 255.)\n#                     x[i] = x[i] / 255.\n\n                # Make a prediction, then take the average.\n                with torch.no_grad():\n                    y_pred = model(x)\n                    y_pred = torch.sigmoid(y_pred.squeeze())\n                    return y_pred[:n].mean().item()\n\n    except Exception as e:\n        print(\"Prediction error on video %s: %s\" % (video_path, str(e)))\n\n    return 0.5","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:56.577554Z","iopub.execute_input":"2024-08-31T16:19:56.577922Z","iopub.status.idle":"2024-08-31T16:19:56.596952Z","shell.execute_reply.started":"2024-08-31T16:19:56.577854Z","shell.execute_reply":"2024-08-31T16:19:56.593734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from concurrent.futures import ThreadPoolExecutor\n\ndef predict_on_video_set(videos, num_workers):\n    def process_file(i):\n        filename = videos[i]\n        y_pred = predict_on_video(os.path.join(test_dir, filename), batch_size=frames_per_video)\n        return y_pred\n\n    with ThreadPoolExecutor(max_workers=num_workers) as ex:\n        predictions = ex.map(process_file, range(len(videos)))\n\n    return list(predictions)","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:56.598695Z","iopub.execute_input":"2024-08-31T16:19:56.599279Z","iopub.status.idle":"2024-08-31T16:19:56.608242Z","shell.execute_reply.started":"2024-08-31T16:19:56.599016Z","shell.execute_reply":"2024-08-31T16:19:56.607021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The leaderboard submission must finish within 9 hours. With 4000 test videos, that is `9*60*60/4000 = 8.1` seconds per video. So if the average time per video is greater than ~8 seconds, the kernel will be too slow!","metadata":{}},{"cell_type":"code","source":"speed_test = False","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:56.609731Z","iopub.execute_input":"2024-08-31T16:19:56.610112Z","iopub.status.idle":"2024-08-31T16:19:56.618100Z","shell.execute_reply.started":"2024-08-31T16:19:56.610040Z","shell.execute_reply":"2024-08-31T16:19:56.617152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if speed_test:\n    start_time = time.time()\n    speedtest_videos = test_videos[:5]\n    predictions = predict_on_video_set(speedtest_videos, num_workers=4)\n    elapsed = time.time() - start_time\n    print(\"Elapsed %f sec. Average per video: %f sec.\" % (elapsed, elapsed / len(speedtest_videos)))","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:56.619389Z","iopub.execute_input":"2024-08-31T16:19:56.619731Z","iopub.status.idle":"2024-08-31T16:19:56.627207Z","shell.execute_reply.started":"2024-08-31T16:19:56.619676Z","shell.execute_reply":"2024-08-31T16:19:56.626179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel.eval()\npredictions = predict_on_video_set(test_videos, num_workers=4)","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:19:56.628726Z","iopub.execute_input":"2024-08-31T16:19:56.629033Z","iopub.status.idle":"2024-08-31T16:29:47.824120Z","shell.execute_reply.started":"2024-08-31T16:19:56.628981Z","shell.execute_reply":"2024-08-31T16:29:47.823222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.DataFrame({\"filename\": test_videos, \"label\": predictions})\nsubmission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:29:47.825634Z","iopub.execute_input":"2024-08-31T16:29:47.825975Z","iopub.status.idle":"2024-08-31T16:29:48.157996Z","shell.execute_reply.started":"2024-08-31T16:29:47.825914Z","shell.execute_reply":"2024-08-31T16:29:48.157346Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:29:48.159372Z","iopub.execute_input":"2024-08-31T16:29:48.159728Z","iopub.status.idle":"2024-08-31T16:29:48.180359Z","shell.execute_reply.started":"2024-08-31T16:29:48.159668Z","shell.execute_reply":"2024-08-31T16:29:48.179636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions[:10]","metadata":{"execution":{"iopub.status.busy":"2024-08-31T16:29:48.199279Z","iopub.execute_input":"2024-08-31T16:29:48.199574Z","iopub.status.idle":"2024-08-31T16:29:48.208233Z","shell.execute_reply.started":"2024-08-31T16:29:48.199520Z","shell.execute_reply":"2024-08-31T16:29:48.207363Z"},"trusted":true},"execution_count":null,"outputs":[]}]}