{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.6.6"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"0aa9b197a8d940bda837f16c4e4f77a1":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":"initial"}},"12a1eeb2a3fb4cb9bc89c5473cdefcc0":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"3d3dbed6c0414f6e9e614d03621305e2":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"55b61a807f784cd1a75657b8fd8672af":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"6cc5e0c3a18342b2b39112289122de3b":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"IntProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"IntProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"100%","description_tooltip":null,"layout":"IPY_MODEL_55b61a807f784cd1a75657b8fd8672af","max":400,"min":0,"orientation":"horizontal","style":"IPY_MODEL_0aa9b197a8d940bda837f16c4e4f77a1","value":400}},"902cf66c7ae349c7b0264d95ec39fede":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_12a1eeb2a3fb4cb9bc89c5473cdefcc0","placeholder":"​","style":"IPY_MODEL_f10eb3086ca1493d979e0d992a85f79a","value":" 400/400 [41:25&lt;00:00,  6.21s/it]"}},"c80cee47cd1e48d8b52090249d2bfca6":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_6cc5e0c3a18342b2b39112289122de3b","IPY_MODEL_902cf66c7ae349c7b0264d95ec39fede"],"layout":"IPY_MODEL_3d3dbed6c0414f6e9e614d03621305e2"}},"f10eb3086ca1493d979e0d992a85f79a":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}}},"version_major":2,"version_minor":0}},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"},{"sourceId":903165,"sourceType":"datasetVersion","datasetId":483879},{"sourceId":1001990,"sourceType":"datasetVersion","datasetId":549794},{"sourceId":1003630,"sourceType":"datasetVersion","datasetId":442595}],"dockerImageVersionId":29845,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# [Task 1] Can you get yourself onboarded with this [Competition](https://www.kaggle.com/competitions/deepfake-detection-challenge)?","metadata":{}},{"cell_type":"markdown","source":"# Deepfake Detection Challenge with Efficient Net\n\n---\n\n## BaseModel:\n\n- Efficientnet-b0(Pretrained)\n\n## Stats:\n\n- Optimizer: Adam\n\n- lr: 0.001\n\n- Schedular: StepLR\n\n- Epochs: 15\n\n- Face Detector: MTCNN\n\n---\n\n## Be careful about:\n\n- Adjust imbalanced data\n\n- Train data is Only 15 images from 1 Movie","metadata":{}},{"cell_type":"markdown","source":"---\n## Library Install","metadata":{}},{"cell_type":"code","source":"%%capture\n# https://www.kaggle.com/timesler/facial-recognition-model-in-pytorch\n# Install facenet-pytorch\n!pip install /kaggle/input/facenet-pytorch-vggface2/facenet_pytorch-1.0.1-py3-none-any.whl\n# Copy model checkpoints to torch cache so they are loaded automatically by the package\n!mkdir -p /tmp/.cache/torch/checkpoints/\n!cp /kaggle/input/facenet-pytorch-vggface2/20180402-114759-vggface2-logits.pth /tmp/.cache/torch/checkpoints/vggface2_DG3kwML46X.pt\n!cp /kaggle/input/facenet-pytorch-vggface2/20180402-114759-vggface2-features.pth /tmp/.cache/torch/checkpoints/vggface2_G5aNV2VSMn.pt","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:10:10.019583Z","iopub.execute_input":"2025-07-06T18:10:10.019775Z","iopub.status.idle":"2025-07-06T18:10:52.814846Z","shell.execute_reply.started":"2025-07-06T18:10:10.019730Z","shell.execute_reply":"2025-07-06T18:10:52.813703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%matplotlib inline\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport random\nimport os\nimport gc\nimport cv2\nimport glob\nimport time\nimport copy\nfrom tqdm.notebook import tqdm\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\n\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, Dataset\nimport torch.nn.functional as F\nimport torchvision\nfrom torchvision import models, transforms\nfrom facenet_pytorch import MTCNN, InceptionResnetV1","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:10:52.817325Z","iopub.execute_input":"2025-07-06T18:10:52.817624Z","iopub.status.idle":"2025-07-06T18:10:56.524495Z","shell.execute_reply.started":"2025-07-06T18:10:52.817572Z","shell.execute_reply":"2025-07-06T18:10:56.523815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sys\npackage_path = '../input/efficientnet-pytorch/EfficientNet-PyTorch/EfficientNet-PyTorch-master'\nsys.path.append(package_path)\n\nfrom efficientnet_pytorch import EfficientNet","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:10:56.528767Z","iopub.execute_input":"2025-07-06T18:10:56.529088Z","iopub.status.idle":"2025-07-06T18:10:56.589323Z","shell.execute_reply.started":"2025-07-06T18:10:56.529027Z","shell.execute_reply":"2025-07-06T18:10:56.588854Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Understand Reproducibility\n- Try running a model training twice without this function — you’ll get slightly different results.\n- Now run with this function called first — you’ll get the same results every time.\n- This is essential for debugging, paper reproduction, and model version control in production.","metadata":{}},{"cell_type":"code","source":"def seed_everything(seed=1234):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:10:56.590184Z","iopub.execute_input":"2025-07-06T18:10:56.590410Z","iopub.status.idle":"2025-07-06T18:10:56.594664Z","shell.execute_reply.started":"2025-07-06T18:10:56.590365Z","shell.execute_reply":"2025-07-06T18:10:56.594018Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"seed_everything(0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:10:56.595746Z","iopub.execute_input":"2025-07-06T18:10:56.595929Z","iopub.status.idle":"2025-07-06T18:10:56.611301Z","shell.execute_reply.started":"2025-07-06T18:10:56.595896Z","shell.execute_reply":"2025-07-06T18:10:56.610751Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## Pretrained Weights","metadata":{}},{"cell_type":"code","source":"# Set Trained Weight Path\nweight_path = 'efficientnet_b0_epoch_15_loss_0.158.pth'\ntrained_weights_path = os.path.join('../input/deepfake-detection-model-weight', weight_path)\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\ntorch.backends.cudnn.benchmark=True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:10:56.612820Z","iopub.execute_input":"2025-07-06T18:10:56.613130Z","iopub.status.idle":"2025-07-06T18:10:56.673702Z","shell.execute_reply.started":"2025-07-06T18:10:56.613078Z","shell.execute_reply":"2025-07-06T18:10:56.673114Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_dir = '../input/deepfake-detection-challenge/test_videos'\nos.listdir(test_dir)[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:10:56.674749Z","iopub.execute_input":"2025-07-06T18:10:56.675017Z","iopub.status.idle":"2025-07-06T18:10:56.706412Z","shell.execute_reply.started":"2025-07-06T18:10:56.674949Z","shell.execute_reply":"2025-07-06T18:10:56.705821Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## Helper function","metadata":{}},{"cell_type":"markdown","source":"Key Concepts\n- Why sample frames? Videos are made up of hundreds/thousands of frames. We don’t need all of them to classify whether it’s a deepfake — sampling a few evenly spaced frames gives a good enough signal with much lower cost.\n- What is frame_window? It’s the interval at which frames are sampled. If frame_window=10, the function picks every 10th frame.\n- Why RGB? OpenCV loads images in BGR format, but most ML libraries (like PyTorch, PIL, etc.) expect RGB — so we convert it.\n\nPlay with Video Sampling\n- Try calling this function on a sample video with num_img=5 and frame_window=10.\n- What happens if you increase or decrease frame_window?\n- Try visualizing the frames using matplotlib.pyplot.imshow(image_list[i]).","metadata":{}},{"cell_type":"code","source":"def get_img_from_mov(video_file, num_img, frame_window):\n    # https://note.nkmk.me/python-opencv-videocapture-file-camera/\n    cap = cv2.VideoCapture(video_file)\n    frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n\n    image_list = []\n    for i in range(num_img):\n        _, image = cap.read()\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        image_list.append(image)\n        cap.set(cv2.CAP_PROP_POS_FRAMES, (i + 1) * frame_window)\n        if cap.get(cv2.CAP_PROP_POS_FRAMES) >= frames:\n            break\n    cap.release()\n\n    return image_list","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:10:56.707739Z","iopub.execute_input":"2025-07-06T18:10:56.708164Z","iopub.status.idle":"2025-07-06T18:10:56.713743Z","shell.execute_reply.started":"2025-07-06T18:10:56.707995Z","shell.execute_reply":"2025-07-06T18:10:56.713134Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Explanation\n\nThis class defines a reusable image preprocessing pipeline, which is a critical step before feeding image data into a neural network.\n\n- `__init__`: The constructor takes three arguments:\n  - `size`: The target height and width to resize the image to (a square of shape `(size, size)`).\n  - `mean`, `std`: These are normalization parameters typically computed from the training dataset. Normalizing input data helps the model converge faster and more reliably.\n\n- `transforms.Compose`: Chains multiple image transformations into one callable pipeline:\n  1. `transforms.Resize((size, size), interpolation=Image.BILINEAR)` resizes the input image to a fixed size using bilinear interpolation (smooth scaling).\n  2. `transforms.ToTensor()` converts the PIL Image or NumPy array to a PyTorch tensor and rescales pixel values from `[0, 255]` to `[0.0, 1.0]`.\n  3. `transforms.Normalize(mean, std)` standardizes the tensor image using the specified per-channel mean and standard deviation.\n\n- `__call__`: Makes the class callable. When an `ImageTransform` instance is called like a function on an image (`transform(img)`), it applies the full sequence of transformations and returns the processed tensor.\n\n---\n\n### Self Thinking\n\n- Why is it beneficial to normalize images with dataset-specific `mean` and `std` rather than using default values?\n- Try changing the resize method from `Image.BILINEAR` to `Image.NEAREST`. What differences do you notice in image quality or model performance?\n- Explore what happens if you skip the normalization step — how does this affect model training or inference stability?\n- Use this class to transform a sample image and inspect the output tensor shape and value range.\n","metadata":{}},{"cell_type":"code","source":"class ImageTransform:\n    def __init__(self, size, mean, std):\n        self.data_transform = transforms.Compose([\n                transforms.Resize((size, size), interpolation=Image.BILINEAR),\n                transforms.ToTensor(),\n                transforms.Normalize(mean, std)\n            ])\n\n    def __call__(self, img):\n        return self.data_transform(img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:10:56.714721Z","iopub.execute_input":"2025-07-06T18:10:56.714960Z","iopub.status.idle":"2025-07-06T18:10:56.727564Z","shell.execute_reply.started":"2025-07-06T18:10:56.714909Z","shell.execute_reply":"2025-07-06T18:10:56.727050Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Explanation\n\nThis class defines a custom `Dataset` for handling deepfake video files. It integrates video decoding, face detection, and image preprocessing in a PyTorch-compatible format for use with `DataLoader`.\n\n- **`__init__`**: Initializes the dataset with:\n  - `file_list`: A list of video file paths.\n  - `device`: The computation device (e.g., `\"cuda\"` or `\"cpu\"`).\n  - `detector`: A face detector object with a `.detect()` method.\n  - `transform`: A transformation function (e.g., an instance of `ImageTransform`) to preprocess images.\n  - `img_num`: The number of frames to extract per video.\n  - `frame_window`: The stride for sampling frames across the video.\n\n- **`__len__`**: Returns the number of samples in the dataset (i.e., the number of videos).\n\n- **`__getitem__`**: \n  - Extracts `img_num` frames from the video using `get_img_from_mov`.\n  - For each frame, runs face detection and crops the face with padding.\n  - Converts the face image to a PIL Image, applies transformation, and appends to the `img_list`.\n  - If face detection fails, it appends `None` and removes them at the end.\n  - Returns the list of processed tensors and the filename.\n\nThis design supports frame sampling, handles noisy or failed detections gracefully, and is optimized for batched processing with `DataLoader`.\n\n---\n\n### Self Thinking\n\n- Why do we need to extract multiple frames from a single video instead of just one?\n- What’s the purpose of padding (±15 pixels) around the detected face region?\n- Think about how you might augment this dataset further (e.g., flip, crop, brightness shift).\n- What are the pros and cons of silently failing (`try/except`) in both frame extraction and face detection? Should we log errors or raise warnings instead?\n- How would this change if multiple faces appear in one frame, or if face detection returns no results?\n","metadata":{}},{"cell_type":"code","source":"class DeepfakeDataset(Dataset):\n    def __init__(self, file_list, device, detector, transform, img_num=20, frame_window=10):\n        self.file_list = file_list\n        self.device = device\n        self.detector = detector\n        self.transform = transform\n        self.img_num = img_num\n        self.frame_window = frame_window\n\n    def __len__(self):\n        return len(self.file_list)\n\n    def __getitem__(self, idx):\n\n        mov_path = self.file_list[idx]\n        img_list = []\n\n        # Movie to Image\n    # try:\n        all_image = get_img_from_mov(mov_path, self.img_num, self.frame_window)\n    # except:\n    #     print(f\"error {idx} file, {mov_path}\")\n    #     return [], mov_path.split('/')[-1]\n\n    \n    # Detect Faces\n        for i,image in enumerate(all_image):\n        \n        # try:\n            _image = image[np.newaxis, :, :, :]\n            boxes, probs = self.detector.detect(_image, landmarks=False)\n            x = int(boxes[0][0][0])\n            y = int(boxes[0][0][1])\n            z = int(boxes[0][0][2])\n            w = int(boxes[0][0][3])\n            image = image[y-15:w+15, x-15:z+15]\n            \n            # Preprocessing\n            image = Image.fromarray(image)\n            image = self.transform(image)\n            \n            img_list.append(image)\n    \n        # except:\n        #     print(f\"error image {image}, i {i}\")\n        #     img_list.append(None)\n            \n        # Padding None\n        img_list = [c for c in img_list if c is not None]\n        \n        return img_list, mov_path.split('/')[-1]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:10:56.728582Z","iopub.execute_input":"2025-07-06T18:10:56.728809Z","iopub.status.idle":"2025-07-06T18:10:56.739754Z","shell.execute_reply.started":"2025-07-06T18:10:56.728766Z","shell.execute_reply":"2025-07-06T18:10:56.739071Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## Model\n\n### Explanation\n\nThis block sets up the model architecture and loads pretrained weights for inference.\n\n- `EfficientNet.from_name('efficientnet-b0')`: Instantiates an EfficientNet-B0 model from the `efficientnet-pytorch` package using a predefined architecture. This base model is often used for image classification tasks due to its balance between accuracy and efficiency.\n\n- `model._fc = nn.Linear(...)`: Replaces the original final classification layer (`_fc`) with a new linear layer:\n  - `in_features`: Uses the number of features output by the backbone network.\n  - `out_features=1`: Specifies a single output node, typically for binary classification (e.g., real vs deepfake).\n\n- `model.load_state_dict(...)`: Loads the saved model parameters (weights) from `trained_weights_path`.\n  - `map_location=torch.device(device)`: Ensures the weights are loaded onto the correct device (`cpu` or `cuda`).\n\nThis setup modifies the EfficientNet model to perform binary classification and restores learned parameters from a previous training run.\n\n---\n\n### Self Thinking\n\n- Why do we replace the `_fc` layer instead of modifying earlier parts of the model?\n- What does `out_features=1` imply about the type of prediction task? How would it change for multi-class classification?\n- Think about what would happen if the architecture defined here doesn’t exactly match the structure used when training the model.\n- What are some alternatives to `torch.load(..., map_location=...)` when deploying to a different device setup?\n","metadata":{}},{"cell_type":"code","source":"model = EfficientNet.from_name('efficientnet-b0')\nmodel._fc = nn.Linear(in_features=model._fc.in_features, out_features=1)\nmodel.load_state_dict(torch.load(trained_weights_path, map_location=torch.device(device)))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:10:56.740536Z","iopub.execute_input":"2025-07-06T18:10:56.740707Z","iopub.status.idle":"2025-07-06T18:11:01.688863Z","shell.execute_reply.started":"2025-07-06T18:10:56.740676Z","shell.execute_reply":"2025-07-06T18:11:01.688185Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# [Task 2] Can you add the training code for [efficientnet model](https://arxiv.org/pdf/1905.11946)?\n\n- Set up your GPU/CPU device detection.\n- Load and Prepare the Dataset\n- Apply preprocessing and augmentations (resize, normalize, crop, flip, etc.).\n- Use DataLoader (PyTorch) or tf.data.Dataset (TensorFlow) to batch and shuffle data.\n- Load the Model from the model zoo (e.g., torchvision.models.efficientnet_b0(pretrained=True)).\n- [key part] Add the last FC layer for this task classification\n- [Key part] Decide whether to fine-tune (train whole model) or transfer learn (freeze backbone and train classifier only).\n- [Key part] Define the Loss Function and Optimizer\n- (Optional) Add a learning rate scheduler.\n- Training Loop For each epoch:\n- Set model to train mode.\n- Loop through batches:\n    - Move data to device (GPU/CPU).\n    - Forward pass: outputs = model(inputs)\n    - Compute loss: loss = criterion(outputs, targets)\n    - Backward pass: loss.backward()\n    - Optimizer step: optimizer.step()\n    - Zero gradients: optimizer.zero_grad()\n- Validation Loop (Optional but Recommended)\n    - Set model to eval mode.\n    - No gradient computation.\n    - Loop through validation data and compute accuracy/loss.\n- Save Checkpoints after each epoch or when performance improves. ('efficientnet_b0_epoch_15_loss_0.158.pth')\n- (Optional) Use TensorBoard, matplotlib, or wandb to track loss, accuracy, etc. for visualization or logging\n","metadata":{}},{"cell_type":"markdown","source":"## Set up your GPU/CPU device detection.","metadata":{}},{"cell_type":"code","source":"device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:11:01.689846Z","iopub.execute_input":"2025-07-06T18:11:01.690088Z","iopub.status.idle":"2025-07-06T18:11:01.693735Z","shell.execute_reply.started":"2025-07-06T18:11:01.690048Z","shell.execute_reply":"2025-07-06T18:11:01.692869Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load and Prepare the Dataset","metadata":{}},{"cell_type":"code","source":"import os\nimport json\nimport cv2\nfrom tqdm import tqdm\nimport torch\nfrom collections import Counter\ndata_dir = '/kaggle/input/deepfake-detection-challenge/train_sample_videos'\nmetadata_path = os.path.join(data_dir, 'metadata.json')\nwith open(metadata_path, 'r') as f:\n    metadata = json.load(f)\n\nfile_list = []\nlabels = []\nfor video_name, meta in metadata.items():\n    video_path = os.path.join(data_dir, video_name)\n    if os.path.exists(video_path):  # 确保视频文件存在\n        file_list.append(video_path)\n        label = 1 if meta['label'] == 'FAKE' else 0  # FAKE为1，REAL为0\n        labels.append(label)\n\nindices = np.arange(len(file_list))\nnp.random.shuffle(indices)\nsplit = int(0.8 * len(indices))  # 80% 训练, 20% 验证\ntrain_idx, val_idx = indices[:split], indices[split:]\n\ntrain_files = [file_list[i] for i in train_idx]\ntrain_labels = [labels[i] for i in train_idx]\nval_files = [file_list[i] for i in val_idx]\nval_labels = [labels[i] for i in val_idx]\nprint(train_files[:10])\nprint(val_files[:10])\n\n# get pos weight\nfrom collections import Counter\nimport torch\nlabel_counter = Counter(labels)\nnum_real = label_counter[0]\nnum_fake = label_counter[1]\n\nprint('real 0:', num_real)\nprint('fake 1:', num_fake)\n# pos_weight_value = num_real / num_fake\n\n# pos_weight = torch.tensor([pos_weight_value], dtype=torch.float32).to(device)\n# print('pos_weight:', pos_weight)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:11:01.696342Z","iopub.execute_input":"2025-07-06T18:11:01.696623Z","iopub.status.idle":"2025-07-06T18:11:03.376887Z","shell.execute_reply.started":"2025-07-06T18:11:01.696572Z","shell.execute_reply":"2025-07-06T18:11:03.376117Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Apply preprocessing and augmentations (resize, normalize, crop, flip, etc.).","metadata":{}},{"cell_type":"code","source":"from torchvision import transforms\nfrom PIL import Image\n\nmean = (0.485, 0.456, 0.406)\nstd = (0.229, 0.224, 0.225)\nimg_size = 224\n\nimport random\n\nclass LabelAwareImageTransform:\n    def __init__(self, size, mean, std, n_fake_aug=3):\n        # real transform\n        self.real_transform = transforms.Compose([\n            transforms.Resize((size, size)),\n            transforms.ToTensor(),\n            transforms.Normalize(mean, std)\n        ])\n        # augmentation\n        self.aug_transform = transforms.Compose([\n            transforms.RandomHorizontalFlip(),\n            transforms.RandomApply([\n                transforms.ColorJitter(0.3, 0.3, 0.3, 0.1)\n            ], p=0.5),\n            transforms.RandomApply([\n                transforms.RandomResizedCrop(size, scale=(0.9, 1.0))\n            ], p=0.7),\n            transforms.Resize((size, size)),\n            transforms.ToTensor(),\n            transforms.Normalize(mean, std)\n        ])\n        self.n_fake_aug = n_fake_aug  # 假样本扩充几份\n\n    def __call__(self, img, label):\n        out = []\n        if label == 0:\n            #real image            \n            out.append(self.real_transform(img))\n            # fake images do augmentation\n            for _ in range(self.n_fake_aug):\n                out.append(self.aug_transform(img))\n        else:\n            out.append(self.real_transform(img))\n        return out\n\ndef collate_fn(batch):\n    imgs, labels = [], []\n    for sample in batch:\n        for img, label in sample:\n            # 强制检查\n            if isinstance(img, torch.Tensor) and img.shape == torch.Size([3, 224, 224]):\n                imgs.append(img)\n                labels.append(label)\n            else:\n                print(\"Skip invalid shape:\", img.shape if isinstance(img, torch.Tensor) else type(img))\n    if len(imgs) == 0:\n        return torch.empty(0), torch.empty(0)\n    imgs = torch.stack(imgs)\n    labels = torch.stack(labels)\n    return imgs, labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:11:06.928195Z","iopub.execute_input":"2025-07-06T18:11:06.928454Z","iopub.status.idle":"2025-07-06T18:11:06.940050Z","shell.execute_reply.started":"2025-07-06T18:11:06.928417Z","shell.execute_reply":"2025-07-06T18:11:06.939057Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Use DataLoader (PyTorch) or tf.data.Dataset (TensorFlow) to batch and shuffle data.","metadata":{}},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import Dataset\nfrom PIL import Image\nimport cv2\nimport numpy as np\n\nclass PairedDeepfakeDataset(Dataset):\n    def __init__(self, file_list, labels, detector, transform, img_num=5, frame_window=10):\n        self.file_list = file_list\n        self.labels = labels\n        self.detector = detector\n        self.transform = transform\n        self.img_num = img_num\n        self.frame_window = frame_window\n\n    def get_img_from_mov(self, video_file, num_img, frame_window):\n        cap = cv2.VideoCapture(video_file)\n        frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n        image_list = []\n        for i in range(num_img):\n            ret, image = cap.read()\n            if not ret or image is None:\n                break\n            image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n            image_list.append(image)\n            cap.set(cv2.CAP_PROP_POS_FRAMES, (i + 1) * frame_window)\n            if cap.get(cv2.CAP_PROP_POS_FRAMES) >= frames:\n                break\n        cap.release()\n        return image_list\n\n    def __len__(self):\n        return len(self.file_list)\n\n    def __getitem__(self, idx):\n        video_path = self.file_list[idx]\n        label = self.labels[idx]\n        # ---- 抽帧 ----\n        frames = self.get_img_from_mov(video_path, self.img_num, self.frame_window)\n        result = []\n        for i,image in enumerate(frames):\n            # try:\n                h, w, _ = image.shape\n                # 人脸检测\n                _image = image[np.newaxis, :, :, :]\n                boxes, probs = self.detector.detect(_image, landmarks=False)\n                if boxes is None or boxes[0] is None or boxes[0][0] is None:\n                    continue\n                x1 = int(max(0, boxes[0][0][0] - 15))\n                y1 = int(max(0, boxes[0][0][1] - 15))\n                x2 = int(min(w, boxes[0][0][2] + 15))\n                y2 = int(min(h, boxes[0][0][3] + 15))\n                crop_img = image[y1:y2, x1:x2]\n                # PIL强制3通道\n                pil_img = Image.fromarray(crop_img).convert('RGB')\n                # ---- label-aware transform，返回list ----\n                tensors = self.transform(pil_img, label)\n                for tensor in tensors:\n                    # if tensor.shape == torch.Size([3, 224, 224]):  # 只收标准尺寸\n                        result.append((tensor, torch.tensor([label], dtype=torch.float32)))\n            # except Exception as e:\n            #     print(f\"image wrong size video_path {video_path} label {label},index {i}\")\n            #     continue\n        return result\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T18:11:15.075720Z","iopub.execute_input":"2025-07-06T18:11:15.076014Z","iopub.status.idle":"2025-07-06T18:11:15.090269Z","shell.execute_reply.started":"2025-07-06T18:11:15.075955Z","shell.execute_reply":"2025-07-06T18:11:15.089396Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# consistent config\nimg_size = 224\nimg_num = 15\nframe_window = 5\nmean = (0.485, 0.456, 0.406)\nstd = (0.229, 0.224, 0.225)\n\ntrain_transform = LabelAwareImageTransform(size=img_size, mean=mean, std=std,n_fake_aug=3)\n# consistent detector\ndetector = MTCNN(\n    image_size=img_size,\n    margin=14,\n    keep_all=False,\n    factor=0.5,\n    select_largest=False,\n    post_process=False,\n    device=device\n)\ntrain_dataset = PairedDeepfakeDataset(\n    file_list=train_files,\n    labels=train_labels,\n    detector=detector,\n    transform=train_transform,\n    img_num=img_num,\n    frame_window=frame_window\n)\n\nval_dataset = PairedDeepfakeDataset(\n    file_list=val_files,\n    labels=val_labels,\n    # device=device,\n    detector=detector,\n    transform=train_transform,  # 可选：验证集只用real_transform\n    img_num=img_num,\n    frame_window=frame_window\n)\n\n\ntrain_loader = DataLoader(\n    train_dataset, \n    batch_size=8,        # 每批处理4个视频，实际图片数=4×img_num\n    shuffle=True, \n    collate_fn=collate_fn, \n    num_workers=0,        # 并行加速，可调大/小\n)\nval_loader = DataLoader(\n    \n    val_dataset, \n    batch_size=8, \n    shuffle=False, \n    collate_fn=collate_fn, \n    num_workers=0)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T20:43:56.332416Z","iopub.execute_input":"2025-07-06T20:43:56.332683Z","iopub.status.idle":"2025-07-06T20:43:56.354576Z","shell.execute_reply.started":"2025-07-06T20:43:56.332646Z","shell.execute_reply":"2025-07-06T20:43:56.354002Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 用于统计\nshape_counter_1 = {}\nshape_counter_0 = {}\nbad_samples = []\n\nfor i in range(len(train_dataset)):\n    sample = train_dataset[i]\n    for img, label in sample:\n        if isinstance(img, torch.Tensor):\n            s = tuple(img.shape)\n            if label == 1:\n                shape_counter_1[s] = shape_counter_1.get(s, 0) + 1\n            else:\n                shape_counter_0[s] = shape_counter_0.get(s, 0) + 1\n            if s != (3, 224, 224):\n                bad_samples.append((i, s, label))\n                print(f\"Bad shape at index {i}: {s}, label={label}\")\n        else:\n            print(f\"Non-tensor at index {i}: {type(img)}\")\n            bad_samples.append((i, type(img), label))\n\nprint(\"Shape distribution:\")\n# print(shape_counter)\nfor k, v in shape_counter_1.items():\n    print(f\" fake {k}: {v} samples\")\nfor k, v in shape_counter_0.items():\n    print(f\" real {k}: {v} samples\")\nprint(f\"Total bad samples: {len(bad_samples)}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T04:12:11.812374Z","iopub.execute_input":"2025-07-06T04:12:11.812683Z","iopub.status.idle":"2025-07-06T04:30:50.688355Z","shell.execute_reply.started":"2025-07-06T04:12:11.812622Z","shell.execute_reply":"2025-07-06T04:30:50.687566Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load the Model from the model zoo (e.g., torchvision.models.efficientnet_b0(pretrained=True)).\n[key part] Add the last FC layer for this task classification.\n\n[Key part] Decide whether to fine-tune (train whole model) or transfer learn (freeze backbone and train classifier only).\n\n[Key part] Define the Loss Function and Optimizer","metadata":{}},{"cell_type":"code","source":"from efficientnet_pytorch import EfficientNet\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.optim.lr_scheduler import StepLR\n\n\n# EfficientNet replace fc\nmodel = EfficientNet.from_name('efficientnet-b0')\nin_features = model._fc.in_features\nmodel._fc = nn.Linear(in_features, 1)  \n\n# load existing weights as pretrain \nweight_path = 'efficientnet_b0_epoch_15_loss_0.158.pth'\ntrained_weights_path = os.path.join('../input/deepfake-detection-model-weight', weight_path)\n\nstate_dict = torch.load(trained_weights_path, map_location=device)\nmodel.load_state_dict(state_dict)\nmodel._fc = nn.Linear(in_features, 1).to(device) #reintialize fc\nmodel = model.to(device)\nprint(state_dict['_fc.weight'].shape)\n\n# freeze backbone\nfor param in model.parameters():\n    param.requires_grad = False\nfor param in model._fc.parameters():\n    param.requires_grad = True\n    \n# optimizer\ncriterion = nn.BCEWithLogitsLoss()\noptimizer = optim.Adam(model._fc.parameters(), lr=0.002)\nscheduler = StepLR(optimizer, step_size=5, gamma=0.1) # decrease lr every 5 epochs\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T20:44:02.756104Z","iopub.execute_input":"2025-07-06T20:44:02.756456Z","iopub.status.idle":"2025-07-06T20:44:02.880913Z","shell.execute_reply.started":"2025-07-06T20:44:02.756405Z","shell.execute_reply":"2025-07-06T20:44:02.880093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install wandb","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T03:17:42.242232Z","iopub.execute_input":"2025-07-06T03:17:42.242495Z","iopub.status.idle":"2025-07-06T03:21:35.959133Z","shell.execute_reply.started":"2025-07-06T03:17:42.242456Z","shell.execute_reply":"2025-07-06T03:21:35.958444Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Training Loop For each epoch:\n\n* Set model to train mode.\n* Loop through batches:Move data to device (GPU/CPU).\n* Forward pass: outputs = model(inputs)\n* Compute loss: loss = criterion(outputs, targets)\n* Backward pass: loss.backward()\n* Optimizer step: optimizer.step()\n* Zero gradients: optimizer.zero_grad() \n## Validation Loop (Optional but Recommended)\n\n* Set model to eval mode.\n* No gradient computation.\n* Loop through validation data and compute accuracy/loss.\n","metadata":{}},{"cell_type":"code","source":"## batch size = 4, lr = 0.001","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\ntrain_losses, val_losses = [], []\ntrain_accs, val_accs = [], []\nnum_epochs=3\nbest_val_loss = float('inf')\nfor epoch in range(num_epochs):\n    # === 训练部分 ===\n    model.train()\n    train_loss, train_correct, train_total = 0, 0, 0\n    for imgs, labels in train_loader:\n        if imgs.nelement() == 0:\n            continue\n        imgs, labels = imgs.to(device), labels.to(device)\n        optimizer.zero_grad()\n        outputs = model(imgs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        preds = (torch.sigmoid(outputs) > 0.5).float()\n        train_correct += (preds == labels).sum().item()\n        train_total += labels.size(0)\n        train_loss += loss.item() * labels.size(0)\n    avg_train_loss = train_loss / train_total if train_total > 0 else 0\n    avg_train_acc = train_correct / train_total if train_total > 0 else 0\n    train_losses.append(avg_train_loss)\n    train_accs.append(avg_train_acc)\n\n    # === 验证部分 ===\n    model.eval()\n    val_loss, val_correct, val_total = 0, 0, 0\n    with torch.no_grad():\n        for imgs, labels in val_loader:\n            if imgs.nelement() == 0:\n                continue\n            imgs, labels = imgs.to(device), labels.to(device)\n            outputs = model(imgs)\n            loss = criterion(outputs, labels)\n            preds = (torch.sigmoid(outputs) > 0.5).float()\n            val_correct += (preds == labels).sum().item()\n            val_total += labels.size(0)\n            val_loss += loss.item() * labels.size(0)\n    avg_val_loss = val_loss / val_total if val_total > 0 else 0\n    avg_val_acc = val_correct / val_total if val_total > 0 else 0\n    val_losses.append(avg_val_loss)\n    val_accs.append(avg_val_acc)\n\n    print(f\"Epoch {epoch+1}/{num_epochs} | \"\n          f\"Train Loss: {avg_train_loss:.4f} | Train Acc: {avg_train_acc:.4f} | \"\n          f\"Val Loss: {avg_val_loss:.4f} | Val Acc: {avg_val_acc:.4f}\")\n\n    if avg_val_loss < best_val_loss:\n        best_val_loss = avg_val_loss\n        torch.save(model.state_dict(), f\"efficientnet_b0_epoch_{epoch+1}_val_loss_{avg_val_loss:.4f}_1.pth\")\n        print(\"Best model saved.\")\n    scheduler.step()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T20:46:47.711573Z","iopub.execute_input":"2025-07-06T20:46:47.711859Z","iopub.status.idle":"2025-07-06T20:47:05.754122Z","shell.execute_reply.started":"2025-07-06T20:46:47.711819Z","shell.execute_reply":"2025-07-06T20:47:05.753026Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# lr = 0.002, epoch=6","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\nimport matplotlib.pyplot as plt\n\ntrain_losses, val_losses = [], []\ntrain_accs, val_accs = [], []\nnum_epochs = 6\nbest_val_loss = float('inf')\n\nfor epoch in range(num_epochs):\n    # === 训练部分 ===\n    model.train()\n    train_loss, train_correct, train_total = 0, 0, 0\n\n    # 加tqdm进度条\n    train_pbar = tqdm(enumerate(train_loader), total=len(train_loader), desc=f\"Epoch {epoch+1} [Train]\")\n    for batch_idx, (imgs, labels) in train_pbar:\n        if imgs.nelement() == 0:\n            continue\n        imgs, labels = imgs.to(device), labels.to(device)\n        optimizer.zero_grad()\n        outputs = model(imgs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        preds = (torch.sigmoid(outputs) > 0.5).float()\n        train_correct += (preds == labels).sum().item()\n        train_total += labels.size(0)\n        train_loss += loss.item() * labels.size(0)\n\n        # 每个batch都更新进度条\n        avg_batch_acc = train_correct / train_total if train_total > 0 else 0\n        train_pbar.set_postfix({\n            \"batch_loss\": f\"{loss.item():.4f}\",\n            \"cumu_acc\": f\"{avg_batch_acc:.4f}\"\n        })\n\n    avg_train_loss = train_loss / train_total if train_total > 0 else 0\n    avg_train_acc = train_correct / train_total if train_total > 0 else 0\n    train_losses.append(avg_train_loss)\n    train_accs.append(avg_train_acc)\n\n    # === 验证部分 ===\n    model.eval()\n    val_loss, val_correct, val_total = 0, 0, 0\n    val_pbar = tqdm(enumerate(val_loader), total=len(val_loader), desc=f\"Epoch {epoch+1} [Val]\")\n    with torch.no_grad():\n        for batch_idx, (imgs, labels) in val_pbar:\n            if imgs.nelement() == 0:\n                continue\n            imgs, labels = imgs.to(device), labels.to(device)\n            outputs = model(imgs)\n            loss = criterion(outputs, labels)\n            preds = (torch.sigmoid(outputs) > 0.5).float()\n            val_correct += (preds == labels).sum().item()\n            val_total += labels.size(0)\n            val_loss += loss.item() * labels.size(0)\n            avg_val_batch_acc = val_correct / val_total if val_total > 0 else 0\n            val_pbar.set_postfix({\n                \"batch_loss\": f\"{loss.item():.4f}\",\n                \"cumu_acc\": f\"{avg_val_batch_acc:.4f}\"\n            })\n\n    avg_val_loss = val_loss / val_total if val_total > 0 else 0\n    avg_val_acc = val_correct / val_total if val_total > 0 else 0\n    val_losses.append(avg_val_loss)\n    val_accs.append(avg_val_acc)\n\n    print(f\"\\nEpoch {epoch+1}/{num_epochs} | \"\n          f\"Train Loss: {avg_train_loss:.4f} | Train Acc: {avg_train_acc:.4f} | \"\n          f\"Val Loss: {avg_val_loss:.4f} | Val Acc: {avg_val_acc:.4f}\")\n\n    if avg_val_loss < best_val_loss:\n        best_val_loss = avg_val_loss\n        torch.save(model.state_dict(), f\"efficientnet_b0_epoch_{epoch+1}_val_loss_{avg_val_loss:.4f}_2.pth\")\n        print(\"Best model saved.\")\n    scheduler.step()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T20:49:59.219292Z","iopub.execute_input":"2025-07-06T20:49:59.219572Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 画 Loss 曲线\nplt.figure(figsize=(8,4))\nplt.plot(train_losses, label='Train Loss')\nplt.plot(val_losses, label='Val Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.title('Training & Validation Loss')\nplt.legend()\nplt.show()\n\n# 画 Accuracy 曲线\nplt.figure(figsize=(8,4))\nplt.plot(train_accs, label='Train Acc')\nplt.plot(val_accs, label='Val Acc')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.title('Training & Validation Accuracy')\nplt.legend()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# [Advanced Task 2] Can you self implement efficient net model using Pytorch?\n\nOverall Structure\n```\nInput: 3 x 224 x 224\n   │\n[Stem] → 3x3 Conv, 32 filters, stride=2\n   ↓\n[Stage 1] → 1 x MBConv1, 16 filters, stride=1\n   ↓\n[Stage 2] → 2 x MBConv6, 24 filters, stride=2 (then 1)\n   ↓\n[Stage 3] → 2 x MBConv6, 40 filters, stride=2 (then 1)\n   ↓\n[Stage 4] → 3 x MBConv6, 80 filters, stride=2 (then 1,1)\n   ↓\n[Stage 5] → 3 x MBConv6, 112 filters, stride=1\n   ↓\n[Stage 6] → 4 x MBConv6, 192 filters, stride=2 (then 1×3)\n   ↓\n[Stage 7] → 1 x MBConv6, 320 filters, stride=1\n   ↓\n[Head] → 1x1 Conv, 1280 filters\n   ↓\nGlobal Average Pooling (1x1)\n   ↓\nDropout (optional)\n   ↓\nFully Connected Layer (num_classes)\n   ↓\nOutput\n```\nMBConv Block\n```\nInput\n  │\n[1x1 Conv] (Expansion, only if expand_ratio > 1)\n  ↓\n[BatchNorm]\n  ↓\n[Swish (SiLU) Activation]\n  ↓\n[Depthwise Conv 3x3 or 5x5] (groups=channels, stride=1 or 2)\n  ↓\n[BatchNorm]\n  ↓\n[Swish]\n  ↓\n[Squeeze-and-Excitation Block]\n  - GlobalAvgPool\n  - FC → ReLU → FC → Sigmoid\n  - Scale the input\n  ↓\n[1x1 Conv] (Projection)\n  ↓\n[BatchNorm]\n  ↓\n[Residual Add] (only if stride==1 and in==out)\n  ↓\nOutput\n```","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nclass SiLU(nn.Module):\n    def forward(self, x):\n        return x * torch.sigmoid(x)\n\nclass SEBlock(nn.Module):\n    def __init__(self, channels, reduction=4):\n        super().__init__()\n        self.fc1 = nn.Conv2d(channels, channels // reduction, 1)\n        self.fc2 = nn.Conv2d(channels // reduction, channels, 1)\n\n    def forward(self, x):\n        s = x.mean((2, 3), keepdim=True)\n        s = F.relu(self.fc1(s))\n        s = torch.sigmoid(self.fc2(s))\n        return x * s\n\nclass MBConv(nn.Module):\n    def __init__(self, in_c, out_c, stride, expand_ratio, kernel_size=3):\n        super().__init__()\n        mid_c = in_c * expand_ratio\n        self.use_res = (stride == 1 and in_c == out_c)\n        layers = []\n        if expand_ratio != 1:\n            # 1x1 expand\n            layers += [\n                nn.Conv2d(in_c, mid_c, 1, bias=False),\n                nn.BatchNorm2d(mid_c),\n                SiLU()\n            ]\n        else:\n            mid_c = in_c\n        # DW conv\n        layers += [\n            nn.Conv2d(mid_c, mid_c, kernel_size, stride, kernel_size // 2, groups=mid_c, bias=False),\n            nn.BatchNorm2d(mid_c),\n            SiLU()\n        ]\n        # SE\n        layers.append(SEBlock(mid_c))\n        # 1x1 project\n        layers += [\n            nn.Conv2d(mid_c, out_c, 1, bias=False),\n            nn.BatchNorm2d(out_c)\n        ]\n        self.block = nn.Sequential(*layers)\n\n    def forward(self, x):\n        out = self.block(x)\n        if self.use_res:\n            return x + out\n        return out\n\nclass EfficientNetB0(nn.Module):\n    def __init__(self, num_classes=1000):\n        super().__init__()\n        # Stem\n        self.stem = nn.Sequential(\n            nn.Conv2d(3, 32, 3, stride=2, padding=1, bias=False),\n            nn.BatchNorm2d(32),\n            SiLU()\n        )\n        # Stage 1: 1 x MBConv1, 16, stride=1\n        self.stage1 = MBConv(32, 16, 1, 1)\n        # Stage 2: 2 x MBConv6, 24, stride=2 (then 1)\n        self.stage2 = nn.Sequential(\n            MBConv(16, 24, 2, 6),\n            MBConv(24, 24, 1, 6)\n        )\n        # Stage 3: 2 x MBConv6, 40, stride=2 (then 1)\n        self.stage3 = nn.Sequential(\n            MBConv(24, 40, 2, 6),\n            MBConv(40, 40, 1, 6)\n        )\n        # Stage 4: 3 x MBConv6, 80, stride=2 (then 1,1)\n        self.stage4 = nn.Sequential(\n            MBConv(40, 80, 2, 6),\n            MBConv(80, 80, 1, 6),\n            MBConv(80, 80, 1, 6)\n        )\n        # Stage 5: 3 x MBConv6, 112, stride=1\n        self.stage5 = nn.Sequential(\n            MBConv(80, 112, 1, 6),\n            MBConv(112, 112, 1, 6),\n            MBConv(112, 112, 1, 6)\n        )\n        # Stage 6: 4 x MBConv6, 192, stride=2 (then 1×3)\n        self.stage6 = nn.Sequential(\n            MBConv(112, 192, 2, 6),\n            MBConv(192, 192, 1, 6),\n            MBConv(192, 192, 1, 6),\n            MBConv(192, 192, 1, 6)\n        )\n        # Stage 7: 1 x MBConv6, 320, stride=1\n        self.stage7 = MBConv(192, 320, 1, 6)\n\n        # Head: 1x1 Conv, 1280\n        self.head = nn.Sequential(\n            nn.Conv2d(320, 1280, 1, bias=False),\n            nn.BatchNorm2d(1280),\n            SiLU()\n        )\n        self.pool = nn.AdaptiveAvgPool2d(1)\n        self.dropout = nn.Dropout(0.2)\n        self.fc = nn.Linear(1280, num_classes)\n\n    def forward(self, x):\n        x = self.stem(x)\n        x = self.stage1(x)\n        x = self.stage2(x)\n        x = self.stage3(x)\n        x = self.stage4(x)\n        x = self.stage5(x)\n        x = self.stage6(x)\n        x = self.stage7(x)\n        x = self.head(x)\n        x = self.pool(x)\n        x = torch.flatten(x, 1)\n        x = self.dropout(x)\n        x = self.fc(x)\n        return x\n\n# 测试shape\nif __name__ == \"__main__\":\n    model = EfficientNetB0(num_classes=1000)\n    print(model)\n    x = torch.randn(2, 3, 224, 224)\n    y = model(x)\n    print(y.shape) # [2, 1000]\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-06T02:57:51.195950Z","iopub.execute_input":"2025-07-06T02:57:51.196282Z","iopub.status.idle":"2025-07-06T02:57:51.632163Z","shell.execute_reply.started":"2025-07-06T02:57:51.196234Z","shell.execute_reply":"2025-07-06T02:57:51.631177Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# [Task 3] Can you leverage Clip embedding here for classification?\n- [Key part] Understand [CLIP](https://arxiv.org/abs/2103.00020)\n- [Key part] Questions for thinking\n    - Q: What's the architecture of Clip?\n \nCLIP has a dual-encoder architecture:\nImage Encoder: Either a ResNet or a Vision Transformer (ViT) that converts an image into a feature vector.\nText Encoder: A Transformer-based model (like GPT) that encodes a text prompt into a vector.\nBoth encoders map their respective inputs into a shared embedding space, where similarity between image and text vectors can be measured via cosine similarity\n      \n    - Q: What's the training objective of Clip?\n \nCLIP has a dual-encoder architecture:\nImage Encoder: Either a ResNet or a Vision Transformer (ViT) that converts an image into a feature vector.\nText Encoder: A Transformer-based model (like GPT) that encodes a text prompt into a vector.\nBoth encoders map their respective inputs into a shared embedding space, where similarity between image and text vectors can be measured via cosine similarity.\n\n    - Q: How does Clip handle image and text embedding alignment?\nCLIP is trained using contrastive learning with a symmetric InfoNCE loss. For each image-text pair in a batch, the model:\nMaximizes the similarity between the correct (matching) pair.\nMinimizes similarity with all incorrect (non-matching) pairs in the batch.\nThe loss is computed from both directions: image-to-text and text-to-image.\nThis trains the model to align image and text embeddings in the shared space.\n    \n    - Q: What's the meaning of \"zero-shot classification\" in Clip?\nZero-shot classification means CLIP can perform image classification without task-specific fine-tuning.\nYou give it text prompts like:\n\"a photo of a cat\", \"a photo of a dog\", ..., one for each class.\nIt encodes the image and each class prompt, computes cosine similarity, and selects the most similar one.\nThis works because CLIP has learned a general image-language understanding, not just class labels.\n    \n    - Q: What's the benifits of contrastive learning vs general classification?\nContrastive learning will learn embedding space that represents semantic similarity while general classification learns decision boundaries. so contrastive learning can be used for retrieval, clustering besides classification task. \nalso it can adapts to different modalities using different models that map to the same embedding space\nit is easy for transfer learning and fine-tune friendly.\n    \n    - Q: What's the limitations of Clip?\nPrompt sensitivity – Performance depends heavily on how you phrase text prompts.\nFine-grained classification is weak – Struggles with subtle differences (e.g., bird species).\nBias from web data – CLIP inherits biases present in large-scale internet data.\nDoesn’t do localization – Can’t tell where an object is in the image.\nFixed encoders – fine-tuning for specific tasks is not easy since you need to align 2 models.\n    \n\n    \n- Advanced Question (better to answer but dont struggle with it)\n    - Q: Why does CLIP use cosine similarity and not Euclidean distance?\nCLIP uses cosine similarity because:\nScale-invariance:\nCosine similarity measures angle between vectors, not their magnitude. This is important because the absolute scale of embeddings doesn't matter, only the direction (semantic content) does.\nStability in high dimensions:\nIn high-dimensional spaces , Euclidean distance becomes less meaningful — all distances tend to become similar. Cosine similarity remains informative.\nNormalization aligns well with contrastive loss:\nCLIP normalizes image and text embeddings to unit vectors, and then computes cosine similarity (which is just a dot product after normalization). This simplifies the contrastive loss and improves numerical stability.\n\n    - Q: How would you adapt CLIP for video understanding?\n    - Q: How is CLIP different from traditional supervised image classifiers like ResNet trained on ImageNet or in our case EfficientNet trained on ImageNet?\n    - Q: How would you evaluate CLIP’s embeddings?","metadata":{}},{"cell_type":"markdown","source":"- Please implement the whole training pipeline to use clip embeddings as input feature instead of efficient net embeddings\n    - import dependency\n    - Load model and implement preprocessing\n    - Freeze Clip Backbone (Optional, for finetuning we should not freeze it but for transfer learning we should)\n    - Define Custom Classifier Head (just like the mlp we did above)\n    - Define Loss, Optimizer, Scheduler\n    - Training Loop\n    - Evaluation Loop\n    - Save Checkpoints\n- Please finishe the rest code to run an inference for this model and get the results as \"submission.csv\"","metadata":{}},{"cell_type":"markdown","source":"# [Task 4] Can you think of ensembling solution for this problem?\n- Think of several backbone embedding generator networks here\n    - Focus on the pros and cons of these embeddings and know why do we want to levereage this\n    - What if the task is a multi modality task (video, audio, text)\n    - When should we do the ensembling?","metadata":{}},{"cell_type":"markdown","source":"---\n## Prediction","metadata":{}},{"cell_type":"code","source":"test_file = [os.path.join(test_dir, path) for path in os.listdir(test_dir)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:52:05.210527Z","iopub.execute_input":"2025-06-07T19:52:05.210809Z","iopub.status.idle":"2025-06-07T19:52:05.215972Z","shell.execute_reply.started":"2025-06-07T19:52:05.210754Z","shell.execute_reply":"2025-06-07T19:52:05.214983Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_file[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:52:05.217435Z","iopub.execute_input":"2025-06-07T19:52:05.217982Z","iopub.status.idle":"2025-06-07T19:52:05.235351Z","shell.execute_reply.started":"2025-06-07T19:52:05.217909Z","shell.execute_reply":"2025-06-07T19:52:05.234564Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Explanation\n\nThis function performs inference on a dataset of deepfake videos using a trained model and returns prediction scores.\n\n- `torch.cuda.empty_cache()`: Clears unused GPU memory to avoid memory accumulation during batch inference.\n\n- `model.eval()`: Sets the model to evaluation mode (important for disabling dropout, batchnorm updates, etc.).\n\n- `with torch.no_grad()`: Turns off gradient tracking to reduce memory usage and improve speed during inference.\n\n- The loop processes each video in the dataset:\n  - `imgs, mov_path = dataset.__getitem__(i)`: Manually retrieves the preprocessed frame tensors and the filename.\n  - If no images were extracted (e.g., face detection failed), assigns a neutral prediction score of `0.5`.\n  - Otherwise, loops over each image (frame), applies the model, and accumulates the sigmoid output (probability).\n  - The final prediction is the **average** prediction across all valid frames from the video.\n\n- Results are stored in `pred_list` and their corresponding video paths in `path_list`.\n\n- The function returns the full list of predicted scores and their associated file names.\n\n---\n\n### Self Thinking\n\n- Why is sigmoid applied to the model output? What does the output represent before and after sigmoid?\n- Why do we average predictions across multiple frames instead of choosing the max, min, or last frame?\n- Consider the implications of defaulting to a `0.5` score for videos where face detection fails — is this the best fallback strategy?\n- What are some ways you might speed up this inference loop, especially for large datasets or when using a GPU?\n- Think about how this function might be modified for multi-class or multi-label tasks.\n","metadata":{}},{"cell_type":"code","source":"# Prediction\ndef predict_dfdc(dataset, model):\n    \n    torch.cuda.empty_cache()\n    pred_list = []\n    path_list = []\n    \n    model = model.to(device)\n    model.eval()\n\n    with torch.no_grad():\n        for i in tqdm(range(len(dataset))):\n            pred = 0\n            imgs, mov_path = dataset.__getitem__(i)\n            \n            # No get Image\n            if len(imgs) == 0:\n                pred_list.append(0.5)\n                path_list.append(mov_path)\n                continue\n                \n                \n            for i in range(len(imgs)):\n                img = imgs[i]\n                \n                output = model(img.unsqueeze(0).to(device))\n                pred += torch.sigmoid(output).item() / len(imgs)\n                \n            pred_list.append(pred)\n            path_list.append(mov_path)\n            \n    torch.cuda.empty_cache()\n            \n    return path_list, pred_list","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:52:05.236733Z","iopub.execute_input":"2025-06-07T19:52:05.237056Z","iopub.status.idle":"2025-06-07T19:52:05.246446Z","shell.execute_reply.started":"2025-06-07T19:52:05.236999Z","shell.execute_reply":"2025-06-07T19:52:05.245373Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Explanation\n\nThis block sets up the configuration and components required for running inference on a set of deepfake video files.\n\n- **Image preprocessing configuration:**\n  - `img_size = 120`: Resizes each detected face to 120x120 pixels.\n  - `img_num = 15`: Number of frames to extract from each video.\n  - `frame_window = 5`: The interval at which frames are sampled.\n  - `mean` and `std`: Normalization statistics used during preprocessing, likely based on ImageNet.\n\n- **`transform = ImageTransform(...)`**: Instantiates the image preprocessing pipeline using the config values.\n\n- **Face detector:**\n  - `MTCNN(...)`: Initializes the [MTCNN](https://kpzhang93.github.io/MTCNN_face_detection_alignment/index.html) face detector with specific hyperparameters:\n    - `margin=14`: Adds a margin around detected faces.\n    - `keep_all=False`: Detects only the most prominent face in each frame.\n    - `select_largest=False`: Avoids favoring the largest face (used in multi-face scenarios).\n    - `post_process=False`: Keeps raw cropped face without alignment.\n    - `.eval()`: Puts the detector into evaluation mode.\n\n- **`dataset = DeepfakeDataset(...)`**: Prepares the inference dataset using all the above components.\n\n- **`predict_dfdc(...)`**: Runs prediction across the dataset, returning:\n  - `path_list`: The filenames of processed videos.\n  - `pred_list`: The predicted probability of each video being a deepfake.\n\nThis setup encapsulates the end-to-end inference pipeline from raw video files to predicted labels.\n\n---\n\n### Self Thinking\n\n- How might changing `img_size` affect model performance and computation time?\n- Why is `keep_all=False` used here? What would happen if you turned it on for videos with multiple faces?\n- If you're deploying this system at scale, what parts of this pipeline would benefit from parallelization or batching?\n- Consider how this config would need to change if using a different model architecture or face detection method.\n- What trade-offs do you notice between choosing more frames (`img_num`) vs faster inference?\n","metadata":{}},{"cell_type":"code","source":"# Config\nimg_size = 120\nimg_num = 15\nframe_window = 5\nmean = (0.485, 0.456, 0.406)\nstd = (0.229, 0.224, 0.225)\n\ntransform = ImageTransform(img_size, mean, std)\n\ndetector = MTCNN(image_size=img_size, margin=14, keep_all=False, factor=0.5, \n                 select_largest=False, post_process=False, device=device).eval()\n\ndataset = DeepfakeDataset(test_file, device, detector, transform, img_num, frame_window)\n\npath_list, pred_list = predict_dfdc(dataset, model)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-06-07T19:52:05.247699Z","iopub.execute_input":"2025-06-07T19:52:05.24792Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---\n## Submission","metadata":{}},{"cell_type":"code","source":"# Submission\nres = pd.DataFrame({\n    'filename': path_list,\n    'label': pred_list,\n})\n\nres.sort_values(by='filename', ascending=True, inplace=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.hist(res['label'], 20)\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"res.head(10)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"res.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}