{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":16880,"databundleVersionId":858837},{"sourceType":"datasetVersion","sourceId":15071268,"datasetId":9649195,"databundleVersionId":15953402},{"sourceType":"kernelVersion","sourceId":41059503}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# --- CHECKING FOR DATASET PATH IN KAGGLE ---\n\nimport os\n\n# This lists exactly what Kaggle named the folders since I don't know in what form it is stored\nprint(\"Searching for your data folders...\")\nfor dirname, _, _ in os.walk('/kaggle/input'):\n    print(f\"Checking path: {dirname}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T17:27:18.401672Z","iopub.execute_input":"2026-03-06T17:27:18.402051Z","iopub.status.idle":"2026-03-06T17:27:25.181972Z","shell.execute_reply.started":"2026-03-06T17:27:18.401912Z","shell.execute_reply":"2026-03-06T17:27:25.18087Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from multiprocessing import Pool, cpu_count","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T14:03:54.604396Z","iopub.execute_input":"2026-03-06T14:03:54.604573Z","iopub.status.idle":"2026-03-06T14:03:54.610578Z","shell.execute_reply.started":"2026-03-06T14:03:54.604541Z","shell.execute_reply":"2026-03-06T14:03:54.609585Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def process_video(row):\n\n    video_path = row[\"video_path\"]\n    label = row[\"label\"]\n\n    frames = extract_frames(video_path)\n\n    data = []\n\n    for frame in frames:\n        data.append((frame, label))\n\n    return data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T14:03:56.703507Z","iopub.execute_input":"2026-03-06T14:03:56.703747Z","iopub.status.idle":"2026-03-06T14:03:56.70908Z","shell.execute_reply.started":"2026-03-06T14:03:56.703708Z","shell.execute_reply":"2026-03-06T14:03:56.707825Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ENTIRE STRUCTURE OF OUR MODEL","metadata":{}},{"cell_type":"code","source":"# deepfake_auditor/\n# │\n# ├── data/\n# │   ├── real/\n# │   └── fake/\n# │\n# ├── frames/\n# │\n# ├── models/\n# │\n# ├── src/\n# │   ├── extract_frames.py\n# │   ├── preprocess_faces.py\n# │   ├── dataset_loader.py\n# │   ├── train_model.py\n# │   ├── evaluate.py\n# │   └── export_model.py\n# │\n# └── main.py","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# video\n# ↓\n# frame extraction\n# ↓\n# face detection\n# ↓\n# face crop\n# ↓\n# RGB image\n# +\n# FFT image\n# ↓\n# CNN model\n# ↓\n# deepfake prediction","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- NECESSARY INSTALLS ---\n\n!pip install tensorflow keras opencv-python numpy pandas scikit-learn matplotlib\n!pip install onnx tf2onnx onnxruntime\n!pip install mtcnn tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T17:35:27.76977Z","iopub.execute_input":"2026-03-06T17:35:27.770026Z","iopub.status.idle":"2026-03-06T17:35:42.454305Z","shell.execute_reply.started":"2026-03-06T17:35:27.769987Z","shell.execute_reply":"2026-03-06T17:35:42.453352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\n\nface_cascade = cv2.CascadeClassifier(\n    cv2.data.haarcascades + 'haarcascade_frontalface_default.xml'\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T18:14:02.122688Z","iopub.execute_input":"2026-03-06T18:14:02.123663Z","iopub.status.idle":"2026-03-06T18:14:02.151495Z","shell.execute_reply.started":"2026-03-06T18:14:02.123621Z","shell.execute_reply":"2026-03-06T18:14:02.15081Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- MAPPING DATASET PATH ---\n\nimport os\n\n# DATASET_PATH = \"/kaggle/input/competitions/deepfake-detection-challenge/train_sample_videos\"\nDATASET_PATHS = [\n\"/kaggle/input/competitions/deepfake-detection-challenge/train_sample_videos\",\n\"/kaggle/input/datasets/akshitrayal/dfdc-train-part-00-to-04/dfdc_train_part_00/dfdc_train_part_0\",\n\"/kaggle/input/datasets/akshitrayal/dfdc-train-part-00-to-04/dfdc_train_part_01/dfdc_train_part_1\",\n\"/kaggle/input/datasets/akshitrayal/dfdc-train-part-00-to-04/dfdc_train_part_02/dfdc_train_part_2\",\n\"/kaggle/input/datasets/akshitrayal/dfdc-train-part-00-to-04/dfdc_train_part_03/dfdc_train_part_3\",\n\"/kaggle/input/datasets/akshitrayal/dfdc-train-part-00-to-04/dfdc_train_part_04/dfdc_train_part_4\",\n]\n\nimport os\n\nvideo_files = []\n\nfor folder in DATASET_PATHS:\n    \n    files = os.listdir(folder)\n\n    for f in files:\n        if f.endswith(\".mp4\"):\n            video_files.append(os.path.join(folder, f))\n\nprint(\"Total videos found:\", len(video_files))\n\nprint(\"\\nFirst 10 videos:\")\nfor v in video_files[:10]:\n    print(v)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T17:59:00.64338Z","iopub.execute_input":"2026-03-06T17:59:00.643773Z","iopub.status.idle":"2026-03-06T17:59:00.844035Z","shell.execute_reply.started":"2026-03-06T17:59:00.643728Z","shell.execute_reply":"2026-03-06T17:59:00.843273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- LOAD DATASET ---\n\nimport json\nimport pandas as pd\n\ndata = []\n\nfor folder in DATASET_PATHS:\n\n    metadata_path = os.path.join(folder, \"metadata.json\")\n\n    if not os.path.exists(metadata_path):\n        continue\n\n    with open(metadata_path) as f:\n        metadata = json.load(f)\n\n    for video, info in metadata.items():\n\n        label = 1 if info[\"label\"] == \"FAKE\" else 0\n\n        video_path = os.path.join(folder, video)\n\n        data.append({\n            \"video_path\": video_path,\n            \"label\": label\n        })\n\ndf = pd.DataFrame(data)\n\nprint(\"Total labeled videos:\", len(df))\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T18:13:45.218098Z","iopub.execute_input":"2026-03-06T18:13:45.218771Z","iopub.status.idle":"2026-03-06T18:13:45.597035Z","shell.execute_reply.started":"2026-03-06T18:13:45.218726Z","shell.execute_reply":"2026-03-06T18:13:45.596461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- BUILDING VIDEO + LABEL TABLE\n\nimport pandas as pd\n\ndata = []\n\nfor video, info in metadata.items():\n\n    video_path = os.path.join(DATASET_PATH, video)\n\n    label = info[\"label\"]\n\n    data.append({\n        \"video_path\": video_path,\n        \"label\": 1 if label == \"FAKE\" else 0\n    })\n\ndf = pd.DataFrame(data)\n\nprint(\"Dataset shape:\", df.shape)\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T08:27:41.674629Z","iopub.execute_input":"2026-03-06T08:27:41.674905Z","iopub.status.idle":"2026-03-06T08:27:42.132386Z","shell.execute_reply.started":"2026-03-06T08:27:41.674864Z","shell.execute_reply":"2026-03-06T08:27:42.131612Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# ASSESING THE DATASET BALANCE\n\nHere the dataset is devided as 80% fake videos which can be a problem if we train directly on this dataset. It may learn bias.\n\n**We will handle this later on with class weights instead pf oversampling**","metadata":{}},{"cell_type":"code","source":"#  --- VERIFYING CLASS BALANCE ---\n\ndf[\"label\"].value_counts()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T08:27:46.154331Z","iopub.execute_input":"2026-03-06T08:27:46.154583Z","iopub.status.idle":"2026-03-06T08:27:46.163467Z","shell.execute_reply.started":"2026-03-06T08:27:46.154546Z","shell.execute_reply":"2026-03-06T08:27:46.162693Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# FRAME EXTRACTION","metadata":{}},{"cell_type":"code","source":"# --- CREATING FRAME STORAGE DIRECTORY ---\n\nimport os\n\nFRAME_DIR = \"/kaggle/working/frames\"\n\nos.makedirs(FRAME_DIR, exist_ok=True)\n\nprint(\"Frame directory created:\", FRAME_DIR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T08:27:52.914753Z","iopub.execute_input":"2026-03-06T08:27:52.915073Z","iopub.status.idle":"2026-03-06T08:27:52.919771Z","shell.execute_reply.started":"2026-03-06T08:27:52.915019Z","shell.execute_reply":"2026-03-06T08:27:52.918931Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- FRAME EXTRACTION FUNCTION WITH FACE CROPPING ---\n\nimport cv2\nimport numpy as np\nfrom tqdm import tqdm\n\ndef extract_frames(video_path, num_frames=30):\n\n    frames = []\n\n    cap = cv2.VideoCapture(video_path)\n\n    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n\n    if total_frames == 0:\n        cap.release()\n        return frames\n\n    frame_indices = np.linspace(0, total_frames-1, num_frames, dtype=int)\n\n    for idx in frame_indices:\n\n        cap.set(cv2.CAP_PROP_POS_FRAMES, idx)\n        success, frame = cap.read()\n\n        if not success:\n            continue\n\n        gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)\n\n        faces = face_cascade.detectMultiScale(\n            gray,\n            scaleFactor=1.3,\n            minNeighbors=5\n        )\n\n        # Take only the first detected face\n        if len(faces) > 0:\n\n            x, y, w, h = faces[0]\n\n            face = frame[y:y+h, x:x+w]\n\n            if face.size == 0:\n                continue\n\n            face = cv2.resize(face, (224,224))\n\n            frames.append(face)\n\n    cap.release()\n\n    return frames","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T08:29:27.49852Z","iopub.execute_input":"2026-03-06T08:29:27.49882Z","iopub.status.idle":"2026-03-06T08:29:27.516428Z","shell.execute_reply.started":"2026-03-06T08:29:27.49878Z","shell.execute_reply":"2026-03-06T08:29:27.515663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- TESTING ---\n\nsample_video = df.iloc[0][\"video_path\"]\n\nframes = extract_frames(sample_video)\n\nprint(\"Extracted frames:\", len(frames))\nprint(\"Frame shape:\", frames[0].shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T10:58:26.277119Z","iopub.execute_input":"2026-03-06T10:58:26.277456Z","iopub.status.idle":"2026-03-06T10:58:26.298521Z","shell.execute_reply.started":"2026-03-06T10:58:26.277409Z","shell.execute_reply":"2026-03-06T10:58:26.297283Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- PARALLEL PROCESSING ---\nrows = df.to_dict(\"records\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T08:36:11.057551Z","iopub.execute_input":"2026-03-06T08:36:11.057878Z","iopub.status.idle":"2026-03-06T08:36:11.063297Z","shell.execute_reply.started":"2026-03-06T08:36:11.057808Z","shell.execute_reply":"2026-03-06T08:36:11.062598Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = []\ny = []\n\nnum_workers = min(4, cpu_count())\n\nwith Pool(num_workers) as pool:\n\n    results = list(tqdm(pool.imap(process_video, rows), total=len(rows)))\n\nfor video_result in results:\n\n    for frame, label in video_result:\n\n        X.append(frame)\n        y.append(label)\n\nprint(\"Total frames collected:\", len(X))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T08:36:30.693198Z","iopub.execute_input":"2026-03-06T08:36:30.693457Z","iopub.status.idle":"2026-03-06T10:17:30.214688Z","shell.execute_reply.started":"2026-03-06T08:36:30.693419Z","shell.execute_reply":"2026-03-06T10:17:30.213783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- CONVERT FRAMES AND LABELS TO NumPy ARRAYS ---\n\nimport numpy as np\n\nX = np.array(X)\ny = np.array(y)\n\nprint(\"X shape:\", X.shape)\nprint(\"y shape:\", y.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T10:55:59.773403Z","iopub.execute_input":"2026-03-06T10:55:59.773744Z","iopub.status.idle":"2026-03-06T10:55:59.778681Z","shell.execute_reply.started":"2026-03-06T10:55:59.773685Z","shell.execute_reply":"2026-03-06T10:55:59.77799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # --- THIS IS FFT IMAGE FEATURE GENERATOR ---\n# # it is creashing the runtime of the environment because it is trying to make a copy of entire dataset in the ram,so I am not going to use it\n\n# import numpy as np\n# import cv2\n\n# def compute_fft_images(images):\n\n#     fft_images = []\n\n#     for img in images:\n\n#         gray = cv2.cvtColor(img.astype(\"uint8\"), cv2.COLOR_BGR2GRAY)\n\n#         f = np.fft.fft2(gray)\n#         fshift = np.fft.fftshift(f)\n\n#         magnitude = 20 * np.log(np.abs(fshift) + 1)\n\n#         magnitude = cv2.resize(magnitude, (224,224))\n\n#         magnitude = np.stack([magnitude]*3, axis=-1)\n\n#         fft_images.append(magnitude)\n\n#     return np.array(fft_images)\n\n\n# print(\"Generating FFT features...\")\n\n# fft_X = compute_fft_images(X)\n\n# X = np.concatenate([X, fft_X], axis=-1)\n\n# print(\"New dataset shape:\", X.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T10:44:46.130403Z","iopub.execute_input":"2026-03-06T10:44:46.130746Z","execution_failed":"2026-03-06T10:46:19.943Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- NORMALIZE PIXELS ---\n# (MIN MAX SCALING)\n\nX = X / 255.0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T10:19:56.259195Z","iopub.execute_input":"2026-03-06T10:19:56.259469Z","iopub.status.idle":"2026-03-06T10:19:59.445727Z","shell.execute_reply.started":"2026-03-06T10:19:56.259431Z","shell.execute_reply":"2026-03-06T10:19:59.445142Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- TRAINING AND VALIDATION SPLIT ---\n\nfrom sklearn.model_selection import train_test_split\n\nX_train, X_val, y_train, y_val = train_test_split(\n    X,\n    y,\n    test_size=0.2,\n    stratify=y,\n    random_state=42\n)\n\nprint(\"Train size:\", X_train.shape)\nprint(\"Validation size:\", X_val.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T10:20:18.013949Z","iopub.execute_input":"2026-03-06T10:20:18.014216Z","iopub.status.idle":"2026-03-06T10:20:22.10629Z","shell.execute_reply.started":"2026-03-06T10:20:18.014179Z","shell.execute_reply":"2026-03-06T10:20:22.105352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\n\nprint(np.unique(y, return_counts=True))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T10:20:41.042142Z","iopub.execute_input":"2026-03-06T10:20:41.042449Z","iopub.status.idle":"2026-03-06T10:20:41.047163Z","shell.execute_reply.started":"2026-03-06T10:20:41.042393Z","shell.execute_reply":"2026-03-06T10:20:41.046465Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# BUILDING THE MODEL","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T10:49:45.177724Z","iopub.execute_input":"2026-03-06T10:49:45.178064Z","iopub.status.idle":"2026-03-06T10:49:49.376238Z","shell.execute_reply.started":"2026-03-06T10:49:45.178016Z","shell.execute_reply":"2026-03-06T10:49:49.375605Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras import layers, models\nfrom tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras.layers import Conv2D, BatchNormalization, GlobalAveragePooling2D, Dense, Dropout, DepthwiseConv2D","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T10:49:49.378153Z","iopub.execute_input":"2026-03-06T10:49:49.378466Z","iopub.status.idle":"2026-03-06T10:49:49.396454Z","shell.execute_reply.started":"2026-03-06T10:49:49.37841Z","shell.execute_reply":"2026-03-06T10:49:49.395946Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# BASE MODEL\nbase_model = MobileNetV2(\n    weights=\"imagenet\",\n    include_top=False,\n    input_shape=(224,224,3)\n)\n\nbase_model.trainable = False\n\nx = base_model.output\n\n# Artifact learning layer\n# x = Conv2D(128, (3,3), padding=\"same\", activation=\"relu\")(x)\n# x = BatchNormalization()(x)\n\n# Deepfake artifact detector\nx = DepthwiseConv2D((3,3), padding=\"same\", activation=\"relu\")(x)\nx = BatchNormalization()(x)\n\nx = Conv2D(128, (1,1), activation=\"relu\")(x)\n\n# pooling\nx = GlobalAveragePooling2D()(x)\n\n# classifier\nx = Dense(256, activation=\"relu\")(x)\nx = Dropout(0.5)(x)\n\noutputs = Dense(1, activation=\"sigmoid\")(x)\n\nmodel = models.Model(inputs=base_model.input, outputs=outputs)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T10:49:49.397395Z","iopub.execute_input":"2026-03-06T10:49:49.397583Z","iopub.status.idle":"2026-03-06T10:49:54.043627Z","shell.execute_reply.started":"2026-03-06T10:49:49.397551Z","shell.execute_reply":"2026-03-06T10:49:54.04272Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- COMPILING THE MODEL ---\n\nmodel.compile(\n    optimizer=\"adam\",\n    loss=\"binary_crossentropy\",\n    metrics=[\"accuracy\"]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T10:49:59.441212Z","iopub.execute_input":"2026-03-06T10:49:59.441494Z","iopub.status.idle":"2026-03-06T10:49:59.491669Z","shell.execute_reply.started":"2026-03-06T10:49:59.441456Z","shell.execute_reply":"2026-03-06T10:49:59.491131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T10:50:00.227432Z","iopub.execute_input":"2026-03-06T10:50:00.227716Z","iopub.status.idle":"2026-03-06T10:50:00.26584Z","shell.execute_reply.started":"2026-03-06T10:50:00.227665Z","shell.execute_reply":"2026-03-06T10:50:00.26506Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- COMPUTING CLASS WEIGHTS ---\n\nfrom sklearn.utils.class_weight import compute_class_weight\nimport numpy as np\n\nclasses = np.unique(y_train)\n\nweights = compute_class_weight(\n    class_weight=\"balanced\",\n    classes=classes,\n    y=y_train\n)\n\nclass_weights = dict(zip(classes, weights))\n\nprint(class_weights)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T10:50:10.076318Z","iopub.execute_input":"2026-03-06T10:50:10.076667Z","iopub.status.idle":"2026-03-06T10:50:10.514193Z","shell.execute_reply.started":"2026-03-06T10:50:10.07661Z","shell.execute_reply":"2026-03-06T10:50:10.513116Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- TRAINING THE MODEL ---\n\nhistory = model.fit(\n    X_train,\n    y_train,\n    validation_data=(X_val, y_val),\n    epochs=10,\n    batch_size=32,\n    class_weight=class_weights\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T10:50:18.516446Z","iopub.execute_input":"2026-03-06T10:50:18.516733Z","iopub.status.idle":"2026-03-06T10:50:18.534788Z","shell.execute_reply.started":"2026-03-06T10:50:18.516686Z","shell.execute_reply":"2026-03-06T10:50:18.533739Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"loss, accuracy = model.evaluate(X_val, y_val)\n\nprint(\"Validation Loss:\", loss)\nprint(\"Validation Accuracy:\", accuracy)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T05:32:29.449387Z","iopub.execute_input":"2026-03-06T05:32:29.449736Z","iopub.status.idle":"2026-03-06T05:32:32.079653Z","shell.execute_reply.started":"2026-03-06T05:32:29.449678Z","shell.execute_reply":"2026-03-06T05:32:32.07861Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"# FINE TUNING THE MODEL\n## UNFREEZING THE LAST 30 LAYERS","metadata":{}},{"cell_type":"code","source":"for layer in base_model.layers[-30:]:\n    layer.trainable = True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T05:36:02.282377Z","iopub.execute_input":"2026-03-06T05:36:02.282686Z","iopub.status.idle":"2026-03-06T05:36:02.287934Z","shell.execute_reply.started":"2026-03-06T05:36:02.282643Z","shell.execute_reply":"2026-03-06T05:36:02.286967Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- RECOMPILING THE MODEL ---\n\nmodel.compile(\n    optimizer=tf.keras.optimizers.Adam(learning_rate=1e-5), # lowering the learning rate\n    loss=\"binary_crossentropy\",\n    metrics=[\"accuracy\"]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T05:36:16.254823Z","iopub.execute_input":"2026-03-06T05:36:16.255091Z","iopub.status.idle":"2026-03-06T05:36:16.387021Z","shell.execute_reply.started":"2026-03-06T05:36:16.25505Z","shell.execute_reply":"2026-03-06T05:36:16.386356Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"We will train for 10 more epoches as the model has already learnt basic features like frame inconsistencies, light distortions, etc which are the basics for identifying a deep fake","metadata":{}},{"cell_type":"code","source":"history_finetune = model.fit(\n    X_train,\n    y_train,\n    validation_data=(X_val, y_val),\n    epochs=10,\n    batch_size=32,\n    class_weight=class_weights\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T05:39:34.348103Z","iopub.execute_input":"2026-03-06T05:39:34.348403Z","iopub.status.idle":"2026-03-06T05:40:37.276966Z","shell.execute_reply.started":"2026-03-06T05:39:34.348357Z","shell.execute_reply":"2026-03-06T05:40:37.276207Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.evaluate(X_val, y_val)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T05:40:45.034501Z","iopub.execute_input":"2026-03-06T05:40:45.034842Z","iopub.status.idle":"2026-03-06T05:40:47.957302Z","shell.execute_reply.started":"2026-03-06T05:40:45.034788Z","shell.execute_reply":"2026-03-06T05:40:47.956563Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# PROBLEM OF LOW ACCURACY AND HIGH LOSS\nReason:\nDeep learning model needs a large dataset and here we only have 3200 frames for training from 320 videos so it learns only basic features from these videos\n","metadata":{}},{"cell_type":"markdown","source":"# IMAGE DATA GENERATOR","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ndatagen = ImageDataGenerator(\n    rotation_range=15,\n    width_shift_range=0.05,\n    height_shift_range=0.05,\n    zoom_range=0.1,\n    horizontal_flip=True,\n    brightness_range=[0.8,1.2]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T05:43:39.970228Z","iopub.execute_input":"2026-03-06T05:43:39.970556Z","iopub.status.idle":"2026-03-06T05:43:39.976994Z","shell.execute_reply.started":"2026-03-06T05:43:39.970508Z","shell.execute_reply":"2026-03-06T05:43:39.976131Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- FITTING THE GENERATOR\n\ndatagen.fit(X_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T05:43:59.881465Z","iopub.execute_input":"2026-03-06T05:43:59.881768Z","iopub.status.idle":"2026-03-06T05:44:01.322784Z","shell.execute_reply.started":"2026-03-06T05:43:59.88172Z","shell.execute_reply":"2026-03-06T05:44:01.321979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history_aug = model.fit(\n    datagen.flow(X_train, y_train, batch_size=32),\n    validation_data=(X_val, y_val),\n    epochs=8,\n    class_weight=class_weights\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-06T05:44:08.202273Z","iopub.execute_input":"2026-03-06T05:44:08.202572Z","iopub.status.idle":"2026-03-06T05:49:23.117332Z","shell.execute_reply.started":"2026-03-06T05:44:08.202525Z","shell.execute_reply":"2026-03-06T05:49:23.116736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}