{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.6"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"0aa9b197a8d940bda837f16c4e4f77a1":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":"initial"}},"12a1eeb2a3fb4cb9bc89c5473cdefcc0":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"3d3dbed6c0414f6e9e614d03621305e2":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"55b61a807f784cd1a75657b8fd8672af":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"6cc5e0c3a18342b2b39112289122de3b":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"IntProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"IntProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"100%","description_tooltip":null,"layout":"IPY_MODEL_55b61a807f784cd1a75657b8fd8672af","max":400,"min":0,"orientation":"horizontal","style":"IPY_MODEL_0aa9b197a8d940bda837f16c4e4f77a1","value":400}},"902cf66c7ae349c7b0264d95ec39fede":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_12a1eeb2a3fb4cb9bc89c5473cdefcc0","placeholder":"​","style":"IPY_MODEL_f10eb3086ca1493d979e0d992a85f79a","value":" 400/400 [41:25&lt;00:00,  6.21s/it]"}},"c80cee47cd1e48d8b52090249d2bfca6":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_6cc5e0c3a18342b2b39112289122de3b","IPY_MODEL_902cf66c7ae349c7b0264d95ec39fede"],"layout":"IPY_MODEL_3d3dbed6c0414f6e9e614d03621305e2"}},"f10eb3086ca1493d979e0d992a85f79a":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}}},"version_major":2,"version_minor":0}},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":903165,"sourceType":"datasetVersion","datasetId":483879},{"sourceId":1001990,"sourceType":"datasetVersion","datasetId":549794},{"sourceId":1003630,"sourceType":"datasetVersion","datasetId":442595}],"dockerImageVersionId":29845,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# [Task 1] Can you get yourself onboarded with this [Competition](https://www.kaggle.com/competitions/deepfake-detection-challenge)?","metadata":{}},{"cell_type":"markdown","source":"# Deepfake Detection Challenge with Efficient Net\n\n---\n\n## BaseModel:\n\n- Efficientnet-b0(Pretrained)\n\n## Stats:\n\n- Optimizer: Adam\n\n- lr: 0.001\n\n- Schedular: StepLR\n\n- Epochs: 15\n\n- Face Detector: MTCNN\n\n---\n\n## Be careful about:\n\n- Adjust imbalanced data\n\n- Train data is Only 15 images from 1 Movie","metadata":{}},{"cell_type":"markdown","source":"---\n## Library Install","metadata":{}},{"cell_type":"code","source":"%%capture\n# https://www.kaggle.com/timesler/facial-recognition-model-in-pytorch\n# Install facenet-pytorch\n!pip install /kaggle/input/facenet-pytorch-vggface2/facenet_pytorch-1.0.1-py3-none-any.whl\n# Copy model checkpoints to torch cache so they are loaded automatically by the package\n!mkdir -p /tmp/.cache/torch/checkpoints/\n!cp /kaggle/input/facenet-pytorch-vggface2/20180402-114759-vggface2-logits.pth /tmp/.cache/torch/checkpoints/vggface2_DG3kwML46X.pt\n!cp /kaggle/input/facenet-pytorch-vggface2/20180402-114759-vggface2-features.pth /tmp/.cache/torch/checkpoints/vggface2_G5aNV2VSMn.pt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:51:11.893219Z","iopub.execute_input":"2025-06-07T19:51:11.893591Z","iopub.status.idle":"2025-06-07T19:51:55.804592Z","shell.execute_reply.started":"2025-06-07T19:51:11.893523Z","shell.execute_reply":"2025-06-07T19:51:55.803698Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%matplotlib inline\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport random\nimport os\nimport gc\nimport cv2\nimport glob\nimport time\nimport copy\nfrom tqdm.notebook import tqdm\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, Dataset\nimport torch.nn.functional as F\nimport torchvision\nfrom torchvision import models, transforms\nfrom facenet_pytorch import MTCNN, InceptionResnetV1","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:51:55.806816Z","iopub.execute_input":"2025-06-07T19:51:55.807103Z","iopub.status.idle":"2025-06-07T19:51:59.859524Z","shell.execute_reply.started":"2025-06-07T19:51:55.807043Z","shell.execute_reply":"2025-06-07T19:51:59.858852Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\npackage_path = '../input/efficientnet-pytorch/EfficientNet-PyTorch/EfficientNet-PyTorch-master'\nsys.path.append(package_path)\n\nfrom efficientnet_pytorch import EfficientNet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:51:59.861069Z","iopub.execute_input":"2025-06-07T19:51:59.861432Z","iopub.status.idle":"2025-06-07T19:51:59.899342Z","shell.execute_reply.started":"2025-06-07T19:51:59.861360Z","shell.execute_reply":"2025-06-07T19:51:59.898534Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Understand Reproducibility\n- Try running a model training twice without this function — you’ll get slightly different results.\n- Now run with this function called first — you’ll get the same results every time.\n- This is essential for debugging, paper reproduction, and model version control in production.","metadata":{}},{"cell_type":"code","source":"def seed_everything(seed=1234):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:51:59.900443Z","iopub.execute_input":"2025-06-07T19:51:59.900677Z","iopub.status.idle":"2025-06-07T19:51:59.905330Z","shell.execute_reply.started":"2025-06-07T19:51:59.900637Z","shell.execute_reply":"2025-06-07T19:51:59.904513Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seed_everything(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:51:59.909642Z","iopub.execute_input":"2025-06-07T19:51:59.909920Z","iopub.status.idle":"2025-06-07T19:51:59.918648Z","shell.execute_reply.started":"2025-06-07T19:51:59.909874Z","shell.execute_reply":"2025-06-07T19:51:59.918022Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## Pretrained Weights","metadata":{}},{"cell_type":"code","source":"# Set Trained Weight Path\nweight_path = 'efficientnet_b0_epoch_15_loss_0.158.pth'\ntrained_weights_path = os.path.join('../input/deepfake-detection-model-weight', weight_path)\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\ntorch.backends.cudnn.benchmark=True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:51:59.920307Z","iopub.execute_input":"2025-06-07T19:51:59.920606Z","iopub.status.idle":"2025-06-07T19:51:59.978892Z","shell.execute_reply.started":"2025-06-07T19:51:59.920535Z","shell.execute_reply":"2025-06-07T19:51:59.977958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_dir = '../input/deepfake-detection-challenge/test_videos'\nos.listdir(test_dir)[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:51:59.980232Z","iopub.execute_input":"2025-06-07T19:51:59.980561Z","iopub.status.idle":"2025-06-07T19:52:00.004994Z","shell.execute_reply.started":"2025-06-07T19:51:59.980503Z","shell.execute_reply":"2025-06-07T19:52:00.004188Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## Helper function","metadata":{}},{"cell_type":"markdown","source":"Key Concepts\n- Why sample frames? Videos are made up of hundreds/thousands of frames. We don’t need all of them to classify whether it’s a deepfake — sampling a few evenly spaced frames gives a good enough signal with much lower cost.\n- What is frame_window? It’s the interval at which frames are sampled. If frame_window=10, the function picks every 10th frame.\n- Why RGB? OpenCV loads images in BGR format, but most ML libraries (like PyTorch, PIL, etc.) expect RGB — so we convert it.\n\nPlay with Video Sampling\n- Try calling this function on a sample video with num_img=5 and frame_window=10.\n- What happens if you increase or decrease frame_window?\n- Try visualizing the frames using matplotlib.pyplot.imshow(image_list[i]).","metadata":{}},{"cell_type":"code","source":"def get_img_from_mov(video_file, num_img, frame_window):\n    # https://note.nkmk.me/python-opencv-videocapture-file-camera/\n    cap = cv2.VideoCapture(video_file)\n    frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n\n    image_list = []\n    for i in range(num_img):\n        _, image = cap.read()\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image_list.append(image)\n        cap.set(cv2.CAP_PROP_POS_FRAMES, (i + 1) * frame_window)\n        if cap.get(cv2.CAP_PROP_POS_FRAMES) >= frames:\n            break\n    cap.release()\n\n    return image_list","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:52:00.006465Z","iopub.execute_input":"2025-06-07T19:52:00.006835Z","iopub.status.idle":"2025-06-07T19:52:00.013056Z","shell.execute_reply.started":"2025-06-07T19:52:00.006731Z","shell.execute_reply":"2025-06-07T19:52:00.012344Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Explanation\n\nThis class defines a reusable image preprocessing pipeline, which is a critical step before feeding image data into a neural network.\n\n- `__init__`: The constructor takes three arguments:\n  - `size`: The target height and width to resize the image to (a square of shape `(size, size)`).\n  - `mean`, `std`: These are normalization parameters typically computed from the training dataset. Normalizing input data helps the model converge faster and more reliably.\n\n- `transforms.Compose`: Chains multiple image transformations into one callable pipeline:\n  1. `transforms.Resize((size, size), interpolation=Image.BILINEAR)` resizes the input image to a fixed size using bilinear interpolation (smooth scaling).\n  2. `transforms.ToTensor()` converts the PIL Image or NumPy array to a PyTorch tensor and rescales pixel values from `[0, 255]` to `[0.0, 1.0]`.\n  3. `transforms.Normalize(mean, std)` standardizes the tensor image using the specified per-channel mean and standard deviation.\n\n- `__call__`: Makes the class callable. When an `ImageTransform` instance is called like a function on an image (`transform(img)`), it applies the full sequence of transformations and returns the processed tensor.\n\n---\n\n### Self Thinking\n\n- Why is it beneficial to normalize images with dataset-specific `mean` and `std` rather than using default values?\n- Try changing the resize method from `Image.BILINEAR` to `Image.NEAREST`. What differences do you notice in image quality or model performance?\n- Explore what happens if you skip the normalization step — how does this affect model training or inference stability?\n- Use this class to transform a sample image and inspect the output tensor shape and value range.\n","metadata":{}},{"cell_type":"code","source":"class ImageTransform:\n    def __init__(self, size, mean, std):\n        self.data_transform = transforms.Compose([\n                transforms.Resize((size, size), interpolation=Image.BILINEAR),\n                transforms.ToTensor(),\n                transforms.Normalize(mean, std)\n            ])\n\n    def __call__(self, img):\n        return self.data_transform(img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:52:00.014489Z","iopub.execute_input":"2025-06-07T19:52:00.014797Z","iopub.status.idle":"2025-06-07T19:52:00.026998Z","shell.execute_reply.started":"2025-06-07T19:52:00.014735Z","shell.execute_reply":"2025-06-07T19:52:00.026093Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Explanation\n\nThis class defines a custom `Dataset` for handling deepfake video files. It integrates video decoding, face detection, and image preprocessing in a PyTorch-compatible format for use with `DataLoader`.\n\n- **`__init__`**: Initializes the dataset with:\n  - `file_list`: A list of video file paths.\n  - `device`: The computation device (e.g., `\"cuda\"` or `\"cpu\"`).\n  - `detector`: A face detector object with a `.detect()` method.\n  - `transform`: A transformation function (e.g., an instance of `ImageTransform`) to preprocess images.\n  - `img_num`: The number of frames to extract per video.\n  - `frame_window`: The stride for sampling frames across the video.\n\n- **`__len__`**: Returns the number of samples in the dataset (i.e., the number of videos).\n\n- **`__getitem__`**: \n  - Extracts `img_num` frames from the video using `get_img_from_mov`.\n  - For each frame, runs face detection and crops the face with padding.\n  - Converts the face image to a PIL Image, applies transformation, and appends to the `img_list`.\n  - If face detection fails, it appends `None` and removes them at the end.\n  - Returns the list of processed tensors and the filename.\n\nThis design supports frame sampling, handles noisy or failed detections gracefully, and is optimized for batched processing with `DataLoader`.\n\n---\n\n### Self Thinking\n\n- Why do we need to extract multiple frames from a single video instead of just one?\n- What’s the purpose of padding (±15 pixels) around the detected face region?\n- Think about how you might augment this dataset further (e.g., flip, crop, brightness shift).\n- What are the pros and cons of silently failing (`try/except`) in both frame extraction and face detection? Should we log errors or raise warnings instead?\n- How would this change if multiple faces appear in one frame, or if face detection returns no results?\n","metadata":{}},{"cell_type":"code","source":"class DeepfakeDataset(Dataset):\n    def __init__(self, file_list, device, detector, transform, img_num=20, frame_window=10):\n        self.file_list = file_list\n        self.device = device\n        self.detector = detector\n        self.transform = transform\n        self.img_num = img_num\n        self.frame_window = frame_window\n\n    def __len__(self):\n        return len(self.file_list)\n\n    def __getitem__(self, idx):\n\n        mov_path = self.file_list[idx]\n        img_list = []\n\n        # Movie to Image\n        try:\n            all_image = get_img_from_mov(mov_path, self.img_num, self.frame_window)\n        except:\n            return [], mov_path.split('/')[-1]\n        \n        # Detect Faces\n        for image in all_image:\n            \n            try:\n                _image = image[np.newaxis, :, :, :]\n                boxes, probs = self.detector.detect(_image, landmarks=False)\n                x = int(boxes[0][0][0])\n                y = int(boxes[0][0][1])\n                z = int(boxes[0][0][2])\n                w = int(boxes[0][0][3])\n                image = image[y-15:w+15, x-15:z+15]\n                \n                # Preprocessing\n                image = Image.fromarray(image)\n                image = self.transform(image)\n                \n                img_list.append(image)\n\n            except:\n                img_list.append(None)\n            \n        # Padding None\n        img_list = [c for c in img_list if c is not None]\n        \n        return img_list, mov_path.split('/')[-1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:52:00.028340Z","iopub.execute_input":"2025-06-07T19:52:00.028683Z","iopub.status.idle":"2025-06-07T19:52:00.041303Z","shell.execute_reply.started":"2025-06-07T19:52:00.028627Z","shell.execute_reply":"2025-06-07T19:52:00.040205Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## Model\n\n### Explanation\n\nThis block sets up the model architecture and loads pretrained weights for inference.\n\n- `EfficientNet.from_name('efficientnet-b0')`: Instantiates an EfficientNet-B0 model from the `efficientnet-pytorch` package using a predefined architecture. This base model is often used for image classification tasks due to its balance between accuracy and efficiency.\n\n- `model._fc = nn.Linear(...)`: Replaces the original final classification layer (`_fc`) with a new linear layer:\n  - `in_features`: Uses the number of features output by the backbone network.\n  - `out_features=1`: Specifies a single output node, typically for binary classification (e.g., real vs deepfake).\n\n- `model.load_state_dict(...)`: Loads the saved model parameters (weights) from `trained_weights_path`.\n  - `map_location=torch.device(device)`: Ensures the weights are loaded onto the correct device (`cpu` or `cuda`).\n\nThis setup modifies the EfficientNet model to perform binary classification and restores learned parameters from a previous training run.\n\n---\n\n### Self Thinking\n\n- Why do we replace the `_fc` layer instead of modifying earlier parts of the model?\n- What does `out_features=1` imply about the type of prediction task? How would it change for multi-class classification?\n- Think about what would happen if the architecture defined here doesn’t exactly match the structure used when training the model.\n- What are some alternatives to `torch.load(..., map_location=...)` when deploying to a different device setup?\n","metadata":{}},{"cell_type":"code","source":"model = EfficientNet.from_name('efficientnet-b0')\nmodel._fc = nn.Linear(in_features=model._fc.in_features, out_features=1)\nmodel.load_state_dict(torch.load(trained_weights_path, map_location=torch.device(device)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:52:00.042586Z","iopub.execute_input":"2025-06-07T19:52:00.042866Z","iopub.status.idle":"2025-06-07T19:52:05.209249Z","shell.execute_reply.started":"2025-06-07T19:52:00.042812Z","shell.execute_reply":"2025-06-07T19:52:05.208165Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# [Task 2] Can you add the training code for [efficientnet model](https://arxiv.org/pdf/1905.11946)?\nOverall Structure\n```\nInput: 3 x 224 x 224\n   │\n[Stem] → 3x3 Conv, 32 filters, stride=2\n   ↓\n[Stage 1] → 1 x MBConv1, 16 filters, stride=1\n   ↓\n[Stage 2] → 2 x MBConv6, 24 filters, stride=2 (then 1)\n   ↓\n[Stage 3] → 2 x MBConv6, 40 filters, stride=2 (then 1)\n   ↓\n[Stage 4] → 3 x MBConv6, 80 filters, stride=2 (then 1,1)\n   ↓\n[Stage 5] → 3 x MBConv6, 112 filters, stride=1\n   ↓\n[Stage 6] → 4 x MBConv6, 192 filters, stride=2 (then 1×3)\n   ↓\n[Stage 7] → 1 x MBConv6, 320 filters, stride=1\n   ↓\n[Head] → 1x1 Conv, 1280 filters\n   ↓\nGlobal Average Pooling (1x1)\n   ↓\nDropout (optional)\n   ↓\nFully Connected Layer (num_classes)\n   ↓\nOutput\n```\nMBConv Block\n```\nInput\n  │\n[1x1 Conv] (Expansion, only if expand_ratio > 1)\n  ↓\n[BatchNorm]\n  ↓\n[Swish (SiLU) Activation]\n  ↓\n[Depthwise Conv 3x3 or 5x5] (groups=channels, stride=1 or 2)\n  ↓\n[BatchNorm]\n  ↓\n[Swish]\n  ↓\n[Squeeze-and-Excitation Block]\n  - GlobalAvgPool\n  - FC → ReLU → FC → Sigmoid\n  - Scale the input\n  ↓\n[1x1 Conv] (Projection)\n  ↓\n[BatchNorm]\n  ↓\n[Residual Add] (only if stride==1 and in==out)\n  ↓\nOutput\n```\n\n# [Advanced Task 2] Can you self implement efficient net model using Pytorch?\n- Set up your GPU/CPU device detection.\n- Load and Prepare the Dataset\n- Apply preprocessing and augmentations (resize, normalize, crop, flip, etc.).\n- Use DataLoader (PyTorch) or tf.data.Dataset (TensorFlow) to batch and shuffle data.\n- Load the Model from the model zoo (e.g., torchvision.models.efficientnet_b0(pretrained=True)).\n- [key part] Add the last FC layer for this task classification\n- [Key part] Decide whether to fine-tune (train whole model) or transfer learn (freeze backbone and train classifier only).\n- [Key part] Define the Loss Function and Optimizer\n- (Optional) Add a learning rate scheduler.\n- Training Loop For each epoch:\n- Set model to train mode.\n- Loop through batches:\n    - Move data to device (GPU/CPU).\n    - Forward pass: outputs = model(inputs)\n    - Compute loss: loss = criterion(outputs, targets)\n    - Backward pass: loss.backward()\n    - Optimizer step: optimizer.step()\n    - Zero gradients: optimizer.zero_grad()\n- Validation Loop (Optional but Recommended)\n    - Set model to eval mode.\n    - No gradient computation.\n    - Loop through validation data and compute accuracy/loss.\n- Save Checkpoints after each epoch or when performance improves. ('efficientnet_b0_epoch_15_loss_0.158.pth')\n- (Optional) Use TensorBoard, matplotlib, or wandb to track loss, accuracy, etc. for visualization or logging\n# [Task 3] Can you leverage Clip embedding here for classification?\n- [Key part] Understand [CLIP](https://arxiv.org/abs/2103.00020)\n- [Key part] Questions for thinking\n    - What's the architecture of Clip?\n    - What's the training objective of Clip?\n    - How does Clip handle image and text embedding alignment?\n    - What's the meaning of \"zero-shot classification\" in Clip?\n    - What's the benifits of contrastive learning vs general classification?\n    - What's the limitations of Clip?\n- Advanced Question (better to answer but dont struggle with it)\n    - Why does CLIP use cosine similarity and not Euclidean distance?\n    - How would you adapt CLIP for video understanding?\n    - How is CLIP different from traditional supervised image classifiers like ResNet trained on ImageNet or in our case EfficientNet trained on ImageNet?\n    - How would you evaluate CLIP’s embeddings?\n- Please implement the whole training pipeline to use clip embeddings as input feature instead of efficient net embeddings\n    - import dependency\n    - Load model and implement preprocessing\n    - Freeze Clip Backbone (Optional, for finetuning we should not freeze it but for transfer learning we should)\n    - Define Custom Classifier Head (just like the mlp we did above)\n    - Define Loss, Optimizer, Scheduler\n    - Training Loop\n    - Evaluation Loop\n    - Save Checkpoints\n- Please finishe the rest code to run an inference for this model and get the results as \"submission.csv\"\n# [Task 4] Can you think of ensembling solution for this problem?\n- Think of several backbone embedding generator networks here\n    - Focus on the pros and cons of these embeddings and know why do we want to levereage this\n    - What if the task is a multi modality task (video, audio, text)\n    - When should we do the ensembling?","metadata":{}},{"cell_type":"markdown","source":"---\n## Prediction","metadata":{}},{"cell_type":"code","source":"test_file = [os.path.join(test_dir, path) for path in os.listdir(test_dir)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:52:05.210527Z","iopub.execute_input":"2025-06-07T19:52:05.210809Z","iopub.status.idle":"2025-06-07T19:52:05.215972Z","shell.execute_reply.started":"2025-06-07T19:52:05.210754Z","shell.execute_reply":"2025-06-07T19:52:05.214983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_file[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:52:05.217435Z","iopub.execute_input":"2025-06-07T19:52:05.217982Z","iopub.status.idle":"2025-06-07T19:52:05.235351Z","shell.execute_reply.started":"2025-06-07T19:52:05.217909Z","shell.execute_reply":"2025-06-07T19:52:05.234564Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Explanation\n\nThis function performs inference on a dataset of deepfake videos using a trained model and returns prediction scores.\n\n- `torch.cuda.empty_cache()`: Clears unused GPU memory to avoid memory accumulation during batch inference.\n\n- `model.eval()`: Sets the model to evaluation mode (important for disabling dropout, batchnorm updates, etc.).\n\n- `with torch.no_grad()`: Turns off gradient tracking to reduce memory usage and improve speed during inference.\n\n- The loop processes each video in the dataset:\n  - `imgs, mov_path = dataset.__getitem__(i)`: Manually retrieves the preprocessed frame tensors and the filename.\n  - If no images were extracted (e.g., face detection failed), assigns a neutral prediction score of `0.5`.\n  - Otherwise, loops over each image (frame), applies the model, and accumulates the sigmoid output (probability).\n  - The final prediction is the **average** prediction across all valid frames from the video.\n\n- Results are stored in `pred_list` and their corresponding video paths in `path_list`.\n\n- The function returns the full list of predicted scores and their associated file names.\n\n---\n\n### Self Thinking\n\n- Why is sigmoid applied to the model output? What does the output represent before and after sigmoid?\n- Why do we average predictions across multiple frames instead of choosing the max, min, or last frame?\n- Consider the implications of defaulting to a `0.5` score for videos where face detection fails — is this the best fallback strategy?\n- What are some ways you might speed up this inference loop, especially for large datasets or when using a GPU?\n- Think about how this function might be modified for multi-class or multi-label tasks.\n","metadata":{}},{"cell_type":"code","source":"# Prediction\ndef predict_dfdc(dataset, model):\n    \n    torch.cuda.empty_cache()\n    pred_list = []\n    path_list = []\n    \n    model = model.to(device)\n    model.eval()\n\n    with torch.no_grad():\n        for i in tqdm(range(len(dataset))):\n            pred = 0\n            imgs, mov_path = dataset.__getitem__(i)\n            \n            # No get Image\n            if len(imgs) == 0:\n                pred_list.append(0.5)\n                path_list.append(mov_path)\n                continue\n                \n                \n            for i in range(len(imgs)):\n                img = imgs[i]\n                \n                output = model(img.unsqueeze(0).to(device))\n                pred += torch.sigmoid(output).item() / len(imgs)\n                \n            pred_list.append(pred)\n            path_list.append(mov_path)\n            \n    torch.cuda.empty_cache()\n            \n    return path_list, pred_list","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:52:05.236733Z","iopub.execute_input":"2025-06-07T19:52:05.237056Z","iopub.status.idle":"2025-06-07T19:52:05.246446Z","shell.execute_reply.started":"2025-06-07T19:52:05.236999Z","shell.execute_reply":"2025-06-07T19:52:05.245373Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Explanation\n\nThis block sets up the configuration and components required for running inference on a set of deepfake video files.\n\n- **Image preprocessing configuration:**\n  - `img_size = 120`: Resizes each detected face to 120x120 pixels.\n  - `img_num = 15`: Number of frames to extract from each video.\n  - `frame_window = 5`: The interval at which frames are sampled.\n  - `mean` and `std`: Normalization statistics used during preprocessing, likely based on ImageNet.\n\n- **`transform = ImageTransform(...)`**: Instantiates the image preprocessing pipeline using the config values.\n\n- **Face detector:**\n  - `MTCNN(...)`: Initializes the [MTCNN](https://kpzhang93.github.io/MTCNN_face_detection_alignment/index.html) face detector with specific hyperparameters:\n    - `margin=14`: Adds a margin around detected faces.\n    - `keep_all=False`: Detects only the most prominent face in each frame.\n    - `select_largest=False`: Avoids favoring the largest face (used in multi-face scenarios).\n    - `post_process=False`: Keeps raw cropped face without alignment.\n    - `.eval()`: Puts the detector into evaluation mode.\n\n- **`dataset = DeepfakeDataset(...)`**: Prepares the inference dataset using all the above components.\n\n- **`predict_dfdc(...)`**: Runs prediction across the dataset, returning:\n  - `path_list`: The filenames of processed videos.\n  - `pred_list`: The predicted probability of each video being a deepfake.\n\nThis setup encapsulates the end-to-end inference pipeline from raw video files to predicted labels.\n\n---\n\n### Self Thinking\n\n- How might changing `img_size` affect model performance and computation time?\n- Why is `keep_all=False` used here? What would happen if you turned it on for videos with multiple faces?\n- If you're deploying this system at scale, what parts of this pipeline would benefit from parallelization or batching?\n- Consider how this config would need to change if using a different model architecture or face detection method.\n- What trade-offs do you notice between choosing more frames (`img_num`) vs faster inference?\n","metadata":{}},{"cell_type":"code","source":"# Config\nimg_size = 120\nimg_num = 15\nframe_window = 5\nmean = (0.485, 0.456, 0.406)\nstd = (0.229, 0.224, 0.225)\n\ntransform = ImageTransform(img_size, mean, std)\n\ndetector = MTCNN(image_size=img_size, margin=14, keep_all=False, factor=0.5, \n                 select_largest=False, post_process=False, device=device).eval()\n\ndataset = DeepfakeDataset(test_file, device, detector, transform, img_num, frame_window)\n\npath_list, pred_list = predict_dfdc(dataset, model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:52:05.247699Z","iopub.execute_input":"2025-06-07T19:52:05.247920Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## Submission","metadata":{}},{"cell_type":"code","source":"# Submission\nres = pd.DataFrame({\n    'filename': path_list,\n    'label': pred_list,\n})\n\nres.sort_values(by='filename', ascending=True, inplace=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.hist(res['label'], 20)\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"res.head(10)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"res.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}