{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":1929873,"sourceType":"datasetVersion","datasetId":1151215},{"sourceId":10125851,"sourceType":"datasetVersion","datasetId":6248577}],"dockerImageVersionId":31193,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Install MediaPipe for face detection\n!pip install mediapipe","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:50:47.318479Z","iopub.execute_input":"2025-11-29T15:50:47.318706Z","iopub.status.idle":"2025-11-29T15:50:50.605213Z","shell.execute_reply.started":"2025-11-29T15:50:47.318682Z","shell.execute_reply":"2025-11-29T15:50:50.604466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport tensorflow as tf\nfrom glob import glob\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import roc_auc_score, average_precision_score\n\nfrom tensorflow.keras.layers import (\n    Dense, Input, Layer, GlobalAveragePooling1D, LayerNormalization, \n    Lambda, Dropout, Reshape, Concatenate\n)\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\n\n# ============================================================\n# 1. CUSTOM LAYERS (Preserved from your working version)\n# ============================================================\n\nclass MultiHeadAttention(Layer):\n    \"\"\"Custom MHA for compatibility with older TF versions.\"\"\"\n    def __init__(self, num_heads, key_dim, **kwargs):\n        super(MultiHeadAttention, self).__init__(**kwargs)\n        self.num_heads = num_heads\n        self.key_dim = key_dim\n        self.d_model = num_heads * key_dim\n        \n        self.query_dense = Dense(self.d_model)\n        self.key_dense = Dense(self.d_model)\n        self.value_dense = Dense(self.d_model)\n        self.combine_heads = Dense(self.d_model)\n\n    def split_heads(self, x, batch_size):\n        x = tf.reshape(x, (batch_size, -1, self.num_heads, self.key_dim))\n        return tf.transpose(x, perm=[0, 2, 1, 3])\n\n    def call(self, query, value, key):\n        batch_size = tf.shape(query)[0]\n        \n        # Linear projections\n        query = self.query_dense(query)\n        key = self.key_dense(key)\n        value = self.value_dense(value)\n        \n        # Split heads\n        query = self.split_heads(query, batch_size)\n        key = self.split_heads(key, batch_size)\n        value = self.split_heads(value, batch_size)\n        \n        # Scaled dot-product attention\n        matmul_qk = tf.matmul(query, key, transpose_b=True)\n        dk = tf.cast(tf.shape(key)[-1], tf.float32)\n        scaled_attention_logits = matmul_qk / tf.math.sqrt(dk)\n        \n        attention_weights = tf.nn.softmax(scaled_attention_logits, axis=-1)\n        output = tf.matmul(attention_weights, value)\n        \n        # Reshape and combine heads\n        output = tf.transpose(output, perm=[0, 2, 1, 3])\n        output = tf.reshape(output, (batch_size, -1, self.d_model))\n        \n        return self.combine_heads(output)\n\n    def get_config(self):\n        config = super(MultiHeadAttention, self).get_config()\n        config.update({\n            \"num_heads\": self.num_heads,\n            \"key_dim\": self.key_dim,\n        })\n        return config\n        \nclass VisionTemporalTransformer(Layer):\n    def __init__(self, patch_size=8, d_model=128, num_heads=4, spatial_layers=1, temporal_layers=1, **kwargs):\n        super(VisionTemporalTransformer, self).__init__(**kwargs)\n        self.patch_size = patch_size\n        self.d_model = d_model\n        self.num_heads = num_heads\n        self.spatial_layers = spatial_layers\n        self.temporal_layers = temporal_layers\n\n        self.dense_projection = Dense(d_model)\n        self.pos_emb = None\n\n        self.spatial_mhas = [MultiHeadAttention(num_heads=num_heads, key_dim=d_model//num_heads) for _ in range(spatial_layers)]\n        self.spatial_norm1 = [LayerNormalization() for _ in range(spatial_layers)]\n        self.spatial_ffn = [tf.keras.Sequential([Dense(d_model*4, activation='relu'), Dense(d_model)]) for _ in range(spatial_layers)]\n        self.spatial_norm2 = [LayerNormalization() for _ in range(spatial_layers)]\n\n        self.temporal_mhas = [MultiHeadAttention(num_heads=num_heads, key_dim=d_model//num_heads) for _ in range(temporal_layers)]\n        self.temporal_norm1 = [LayerNormalization() for _ in range(temporal_layers)]\n        self.temporal_ffn = [tf.keras.Sequential([Dense(d_model*4, activation='relu'), Dense(d_model)]) for _ in range(temporal_layers)]\n        self.temporal_norm2 = [LayerNormalization() for _ in range(temporal_layers)]\n        \n    def build(self, input_shape):\n        H = input_shape[2]\n        W = input_shape[3]\n        ph = H // self.patch_size\n        pw = W // self.patch_size\n        num_patches = ph * pw\n        self.pos_emb = self.add_weight(shape=(1, num_patches, self.d_model), initializer='random_normal', trainable=True, name='pos_emb')\n        super(VisionTemporalTransformer, self).build(input_shape)\n\n    def call(self, inputs):\n        # 1. Handle Shapes\n        input_shape = inputs.get_shape() \n        shape = tf.shape(inputs)\n        \n        batch = shape[0]\n        frames = shape[1]\n        H = shape[2]\n        W = shape[3]\n        \n        # Attempt to get static Channel dim\n        C_static = input_shape[-1]\n        C = C_static if C_static is not None else shape[4]\n\n        # 2. Reshape\n        reshaped = tf.reshape(inputs, (-1, H, W, C))\n\n        # 3. Extract Patches\n        patches = tf.image.extract_patches(\n            images=reshaped,\n            sizes=[1, self.patch_size, self.patch_size, 1],\n            strides=[1, self.patch_size, self.patch_size, 1],\n            rates=[1,1,1,1],\n            padding='VALID'\n        )\n        \n        # 4. Flatten Patches & FORCE SHAPE\n        if C_static is not None:\n            patch_dim_static = self.patch_size * self.patch_size * C_static\n        else:\n            patch_dim_static = None\n            \n        patch_dim_dynamic = tf.shape(patches)[-1]\n        final_patch_dim = patch_dim_static if patch_dim_static is not None else patch_dim_dynamic\n        \n        patches = tf.reshape(patches, (-1, tf.shape(patches)[1] * tf.shape(patches)[2], final_patch_dim))\n\n        if patch_dim_static is not None:\n             patches.set_shape([None, None, patch_dim_static])\n\n        # 5. Projection\n        x = self.dense_projection(patches) + self.pos_emb\n\n        # 6. Spatial Transformer\n        for i in range(self.spatial_layers):\n            attn = self.spatial_mhas[i](x, value=x, key=x)\n            x = self.spatial_norm1[i](x + attn)\n            ff = self.spatial_ffn[i](x)\n            x = self.spatial_norm2[i](x + ff)\n\n        # 7. Temporal Pooling\n        x = tf.reshape(x, (batch, frames, -1, self.d_model))\n        x = tf.reduce_mean(x, axis=2)  \n\n        # FORCE STATIC SHAPE FOR TEMPORAL (Required for older TF)\n        x.set_shape([None, None, self.d_model]) \n\n        # 8. Temporal Transformer\n        for i in range(self.temporal_layers):\n            attn = self.temporal_mhas[i](x, value=x, key=x)\n            x = self.temporal_norm1[i](x + attn)\n            ff = self.temporal_ffn[i](x)\n            x = self.temporal_norm2[i](x + ff)\n\n        pooled = GlobalAveragePooling1D()(x)\n        return pooled\n\n    def get_config(self):\n        config = super(VisionTemporalTransformer, self).get_config()\n        config.update({\n            \"patch_size\": self.patch_size,\n            \"d_model\": self.d_model,\n            \"num_heads\": self.num_heads,\n            \"spatial_layers\": self.spatial_layers,\n            \"temporal_layers\": self.temporal_layers,\n        })\n        return config\n\n# ============================================================\n# 2. MODEL DEFINITION\n# ============================================================\n\ndef batch_consistency_loss(y_true, features):\n    f = tf.reshape(features, (tf.shape(features)[0], -1))\n    f_norm = tf.math.l2_normalize(f, axis=1)\n    sim_matrix = tf.matmul(f_norm, f_norm, transpose_b=True)\n    avg_sim = tf.reduce_mean(sim_matrix, axis=1)\n    return 1.0 - avg_sim\n\ndef consistency_loss_wrapper(y_true, y_pred):\n    return batch_consistency_loss(y_true, y_pred)\n\ndef build_lipinc_model(frame_shape=(8,64,144,3), residue_shape=(7,64,144,3), d_model=128):\n    frame_input = Input(shape=frame_shape, name='FrameInput')\n    residue_input = Input(shape=residue_shape, name='ResidueInput')\n\n    vt = VisionTemporalTransformer(\n        patch_size=8, d_model=d_model, num_heads=4, spatial_layers=1, temporal_layers=1\n    )\n\n    frame_feat = vt(frame_input)      \n    residue_feat = vt(residue_input) \n\n    expand1 = Lambda(lambda x: tf.expand_dims(x, axis=1))\n    q = expand1(frame_feat)\n    k = expand1(residue_feat)\n    v = k\n\n    mha = MultiHeadAttention(num_heads=4, key_dim=d_model//4)\n    attn_out = mha(q, value=v, key=k)  \n\n    squeeze = Lambda(lambda x: tf.squeeze(x, axis=1))\n    attn_out = squeeze(attn_out)\n\n    concat = Lambda(lambda t: tf.concat(t, axis=1))\n    fusion = concat([frame_feat, residue_feat, attn_out])\n\n    x = Dense(512, activation='relu')(fusion)\n    x = Dense(256, activation='relu')(x)\n\n    class_output = Dense(2, activation='softmax', name='class_output')(x)\n    features_output = Dense(d_model, activation=None, name='features_output')(x)\n\n    model = Model(\n        inputs=[frame_input, residue_input],\n        outputs=[class_output, features_output],\n        name='LIPINC_fixed'\n    )\n    return model\n\nprint(\"Model architecture loaded.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:50:50.607612Z","iopub.execute_input":"2025-11-29T15:50:50.607893Z","iopub.status.idle":"2025-11-29T15:50:56.088051Z","shell.execute_reply.started":"2025-11-29T15:50:50.607860Z","shell.execute_reply":"2025-11-29T15:50:56.087203Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport numpy as np\nimport mediapipe as mp\n\nclass MouthExtractor:\n    def __init__(self):\n        self.mp_face_mesh = mp.solutions.face_mesh\n        # ref: https://github.com/google/mediapipe/blob/master/mediapipe/python/solutions/face_mesh.py\n        self.face_mesh = self.mp_face_mesh.FaceMesh(\n            static_image_mode=False,\n            max_num_faces=1,\n            refine_landmarks=True, # Improves lip accuracy\n            min_detection_confidence=0.5\n        )\n        print(\"✅ MediaPipe FaceMesh Loaded (No Dlib required!)\")\n\n        # MediaPipe Indices for Lips (Outer and Inner)\n        # We use these to calculate the bounding box\n        self.Mouth_Indices = [\n            61, 146, 91, 181, 84, 17, 314, 405, 321, 375, 291, # Upper Lip\n            61, 185, 40, 39, 37, 0, 267, 269, 270, 409, 291    # Lower Lip\n        ]\n        \n        # Indices for calculating openness (Upper Inner vs Lower Inner)\n        self.upper_inner = 13\n        self.lower_inner = 14\n        self.mouth_left = 61\n        self.mouth_right = 291\n\n    def get_mouth_openness(self, landmarks, img_w, img_h):\n        # Convert normalized coordinates to pixels\n        u = landmarks[self.upper_inner]\n        l = landmarks[self.lower_inner]\n        left = landmarks[self.mouth_left]\n        right = landmarks[self.mouth_right]\n\n        # Calculate height and width\n        height = np.linalg.norm(np.array([u.x * img_w, u.y * img_h]) - np.array([l.x * img_w, l.y * img_h]))\n        width = np.linalg.norm(np.array([left.x * img_w, left.y * img_h]) - np.array([right.x * img_w, right.y * img_h]))\n\n        return height / (width + 1e-6)\n\n    def crop_mouth(self, img, landmarks):\n        h, w, _ = img.shape\n        \n        # Get all mouth landmark coordinates\n        mouth_points = []\n        for idx in self.Mouth_Indices:\n            pt = landmarks[idx]\n            mouth_points.append([int(pt.x * w), int(pt.y * h)])\n        \n        mouth_points = np.array(mouth_points)\n        \n        # Bounding Box\n        x, y, mw, mh = cv2.boundingRect(mouth_points)\n        \n        # Add Padding (20% width, 50% height)\n        pad_w = int(mw * 0.2)\n        pad_h = int(mh * 0.5)\n        \n        x1 = max(0, x - pad_w)\n        y1 = max(0, y - pad_h)\n        x2 = min(w, x + mw + pad_w)\n        y2 = min(h, y + mh + pad_h)\n        \n        crop = img[y1:y2, x1:x2]\n        \n        # Resize to (144, 64) -> CV2 expects (W, H)\n        resized = cv2.resize(crop, (144, 64))\n        return resized\n\n    def process_video(self, video_path, frame_count=8, scan_limit=60):\n        cap = cv2.VideoCapture(video_path)\n        frames_buffer = []\n        openness_scores = []\n        crops_buffer = []\n        \n        count = 0\n        while True:\n            ret, frame = cap.read()\n            if not ret or count >= scan_limit:\n                break\n            \n            # Convert BGR to RGB for MediaPipe\n            rgb_frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n            results = self.face_mesh.process(rgb_frame)\n            \n            if results.multi_face_landmarks:\n                landmarks = results.multi_face_landmarks[0].landmark\n                h, w, _ = frame.shape\n                \n                score = self.get_mouth_openness(landmarks, w, h)\n                crop = self.crop_mouth(frame, landmarks)\n                \n                openness_scores.append((count, score))\n                crops_buffer.append(crop)\n                frames_buffer.append(frame) # Keep track of valid frames\n            \n            count += 1\n        cap.release()\n\n        # Fallback if no faces found\n        if len(crops_buffer) < frame_count:\n            return self._fallback_processing(frames_buffer if frames_buffer else [np.zeros((64,144,3), dtype=np.uint8)], frame_count)\n\n        # Selection Logic: Prioritize High Openness\n        openness_scores.sort(key=lambda x: x[1], reverse=True)\n        \n        # Pick indices evenly spaced from the valid detections\n        indices = np.linspace(0, len(crops_buffer)-1, frame_count, dtype=int)\n        \n        # Or, if you want strictly the most open mouths, use the top N scores\n        # But for video consistency, temporal spacing is usually better.\n        final_crops = [crops_buffer[i] for i in indices]\n        \n        return np.array(final_crops) / 255.0\n\n    def _fallback_processing(self, frames, target_count):\n        processed = []\n        # If we have frames but no face, just resize the whole center\n        for i in range(target_count):\n            if i < len(frames):\n                resized = cv2.resize(frames[i], (144, 64))\n                processed.append(resized)\n            else:\n                processed.append(np.zeros((64, 144, 3)))\n        return np.array(processed) / 255.0\n\n# Initialize global extractor\nmouth_extractor = MouthExtractor()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:50:56.088806Z","iopub.execute_input":"2025-11-29T15:50:56.089251Z","iopub.status.idle":"2025-11-29T15:50:56.260184Z","shell.execute_reply.started":"2025-11-29T15:50:56.089230Z","shell.execute_reply":"2025-11-29T15:50:56.259221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class VideoDataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, video_paths, labels, batch_size=16, frame_count=8, dim=(64, 144), shuffle=True, **kwargs):\n        # FIX 1: Pass kwargs to super for multiprocessing support\n        super().__init__(**kwargs)\n        \n        self.video_paths = video_paths\n        self.labels = labels\n        self.batch_size = batch_size\n        self.frame_count = frame_count\n        self.dim = dim\n        self.shuffle = shuffle\n        self.indexes = np.arange(len(self.video_paths))\n        self.on_epoch_end()\n\n    def __len__(self):\n        return int(np.floor(len(self.video_paths) / self.batch_size))\n\n    def __getitem__(self, index):\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        list_paths = [self.video_paths[k] for k in indexes]\n        list_labels = [self.labels[k] for k in indexes]\n        return self.__data_generation(list_paths, list_labels)\n\n    def on_epoch_end(self):\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n\n    def compute_residue(self, frames):\n        # Section C: Delta Frames (Dt = Rt+1 - Rt)\n        residues = np.zeros((self.frame_count - 1, self.dim[0], self.dim[1], 3), dtype=np.float32)\n        for i in range(1, len(frames)):\n            residues[i-1] = frames[i] - frames[i-1]\n        return residues\n\n    def __data_generation(self, list_paths, list_labels):\n        X_frames = np.empty((self.batch_size, self.frame_count, *self.dim, 3))\n        X_residues = np.empty((self.batch_size, self.frame_count-1, *self.dim, 3))\n        y = np.empty((self.batch_size, 2), dtype=int)\n        \n        # Dummy features target for the consistency loss\n        dummy_feats = np.zeros((self.batch_size, 128))\n\n        for i, path in enumerate(list_paths):\n            # USE MOUTH EXTRACTOR (Assumes mouth_extractor is defined globally)\n            frames = mouth_extractor.process_video(path, frame_count=self.frame_count)\n            \n            # Ensure shape consistency\n            if frames.shape != (self.frame_count, self.dim[0], self.dim[1], 3):\n                # Emergency pad/resize if something went wrong\n                if len(frames) > 0:\n                    frames = cv2.resize(frames[0], (self.dim[1], self.dim[0]))\n                    frames = np.array([frames] * self.frame_count)\n                else:\n                    frames = np.zeros((self.frame_count, self.dim[0], self.dim[1], 3))\n\n            X_frames[i,] = frames\n            X_residues[i,] = self.compute_residue(frames)\n            y[i] = list_labels[i]\n\n        # FIX 2: Return Dictionaries instead of Lists\n        # Matches Input Layer names: 'FrameInput', 'ResidueInput'\n        X = {\n            'FrameInput': X_frames,\n            'ResidueInput': X_residues\n        }\n        \n        # Matches Output Layer names\n        targets = {\n            'class_output': y, \n            'features_output': dummy_feats\n        }\n\n        return X, targets","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:50:56.261294Z","iopub.execute_input":"2025-11-29T15:50:56.261615Z","iopub.status.idle":"2025-11-29T15:50:56.273695Z","shell.execute_reply.started":"2025-11-29T15:50:56.261587Z","shell.execute_reply":"2025-11-29T15:50:56.272814Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport json\nimport numpy as np\nimport os\nfrom glob import glob\nfrom sklearn.model_selection import train_test_split\n\n# -----------------------------------------------------------\n# 1. SETUP PATHS (FaceForensics++ Only)\n# -----------------------------------------------------------\nFF_REAL_PATH = \"/kaggle/input/ff-c23/FaceForensics++_C23/original\"\n\n# List of all Fake sub-datasets in FaceForensics++\nFF_FAKE_PATHS = [\n    \"/kaggle/input/ff-c23/FaceForensics++_C23/Deepfakes\",\n    \"/kaggle/input/ff-c23/FaceForensics++_C23/Face2Face\",\n    \"/kaggle/input/ff-c23/FaceForensics++_C23/FaceSwap\",\n    \"/kaggle/input/ff-c23/FaceForensics++_C23/NeuralTextures\"\n]\n\n# -----------------------------------------------------------\n# 2. LOAD REAL VIDEOS\n# -----------------------------------------------------------\nprint(\"Scanning FaceForensics++ Real Videos...\")\nff_real = glob(os.path.join(FF_REAL_PATH, '**', '*.mp4'), recursive=True)\n\n# Fallback for .avi if .mp4 is empty (Dataset specific check)\nif not ff_real: \n    ff_real = glob(os.path.join(FF_REAL_PATH, '**', '*.avi'), recursive=True)\n\nprint(f\"Real Videos found: {len(ff_real)}\")\n\n# -----------------------------------------------------------\n# 3. LOAD FAKE VIDEOS (Aggregating all 4 types)\n# -----------------------------------------------------------\nprint(\"\\nScanning FaceForensics++ Fake Videos (Deepfakes, F2F, FaceSwap, NeuralTextures)...\")\nff_fake = []\n\nfor fake_path in FF_FAKE_PATHS:\n    # Recursively find videos in each sub-folder\n    videos = glob(os.path.join(fake_path, '**', '*.mp4'), recursive=True)\n    if not videos:\n        videos = glob(os.path.join(fake_path, '**', '*.avi'), recursive=True)\n    \n    print(f\" - Found {len(videos)} in {os.path.basename(fake_path)}\")\n    ff_fake.extend(videos)\n\nprint(f\"Total Fake Videos found: {len(ff_fake)}\")\n\n# -----------------------------------------------------------\n# 4. BALANCE DATA (Random Undersampling)\n# -----------------------------------------------------------\nprint(\"\\n--- Balancing Data ---\")\n\n# Convert to numpy array for easier indexing\nff_fake = np.array(ff_fake)\nff_real = np.array(ff_real)\n\nn_real = len(ff_real)\nn_fake = len(ff_fake)\n\nprint(f\"Counts -> Real: {n_real}, Fake: {n_fake}\")\n\nif n_fake > n_real:\n    print(f\"Undersampling Fake videos from {n_fake} to {n_real}...\")\n    # Randomly choose 'n_real' indices from fake paths without replacement\n    ff_fake_balanced = np.random.choice(ff_fake, n_real, replace=False)\nelse:\n    print(\"Fake videos are fewer or equal to Real. No undersampling needed.\")\n    ff_fake_balanced = ff_fake\n\n# Final Lists used for training\nfinal_real_paths = ff_real\nfinal_fake_paths = ff_fake_balanced\n\n# Create Labels: [1, 0] for Real, [0, 1] for Fake\nfinal_real_labels = [[1, 0]] * len(final_real_paths)\nfinal_fake_labels = [[0, 1]] * len(final_fake_paths)\n\n# Merge into single dataset\nall_paths = np.concatenate([final_real_paths, final_fake_paths])\nall_labels = np.concatenate([final_real_labels, final_fake_labels])\n\nprint(f\"Final Balanced Dataset: {len(all_paths)} total videos ({len(final_real_paths)} Real, {len(final_fake_paths)} Fake)\")\n\n# -----------------------------------------------------------\n# 5. SPLIT DATA\n# -----------------------------------------------------------\n# Stratified split to maintain class balance in Train/Val/Test\nX_train_paths, X_temp_paths, y_train, y_temp = train_test_split(\n    all_paths, all_labels, test_size=0.3, random_state=42, stratify=all_labels\n)\n\nX_val_paths, X_test_paths, y_val, y_test = train_test_split(\n    X_temp_paths, y_temp, test_size=0.5, random_state=42, stratify=y_temp\n)\n\nprint(f\"\\nTraining on: {len(X_train_paths)}\")\nprint(f\"Validation on: {len(X_val_paths)}\")\nprint(f\"Testing on: {len(X_test_paths)}\")\n\n# -----------------------------------------------------------\n# 6. INSTANTIATE GENERATORS\n# -----------------------------------------------------------\nBATCH_SIZE = 16\n\ntrain_gen = VideoDataGenerator(X_train_paths, y_train, batch_size=BATCH_SIZE)\nval_gen = VideoDataGenerator(X_val_paths, y_val, batch_size=BATCH_SIZE)\ntest_gen = VideoDataGenerator(X_test_paths, y_test, batch_size=BATCH_SIZE, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:50:56.274593Z","iopub.execute_input":"2025-11-29T15:50:56.274884Z","iopub.status.idle":"2025-11-29T15:51:08.569648Z","shell.execute_reply.started":"2025-11-29T15:50:56.274859Z","shell.execute_reply":"2025-11-29T15:51:08.568988Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tensorflow.keras.callbacks import ModelCheckpoint, EarlyStopping\nfrom tensorflow.keras.optimizers import Adam\n\n# -----------------------------------------------------------\n# 1. BUILD & COMPILE\n# -----------------------------------------------------------\nmodel = build_lipinc_model()\nmodel.summary()\n\nopt = Adam(learning_rate=1e-4)\n\nmodel.compile(\n    optimizer=opt,\n    loss={\n        # Naming these keys exactly creates the \"class_output_loss\" log output\n        'class_output': 'categorical_crossentropy', \n        'features_output': consistency_loss_wrapper \n    },\n    loss_weights={\n        'class_output': 1.0, \n        'features_output': 5.0 \n    },\n    metrics={'class_output': 'accuracy'}\n)\n\n# -----------------------------------------------------------\n# 2. DEFINE CALLBACKS\n# -----------------------------------------------------------\ncheckpoint = ModelCheckpoint(\n    'best_lipinc_model.h5', \n    monitor='val_class_output_accuracy', \n    save_best_only=True, \n    mode='max',\n    verbose=1  # <--- CRITICAL: This enables the \"Epoch 01: saving model...\" printout\n)\n\nearly_stop = EarlyStopping(\n    monitor='val_loss', \n    patience=8, \n    restore_best_weights=True,\n    verbose=1\n)\n\n# -----------------------------------------------------------\n# 3. TRAIN\n# -----------------------------------------------------------\nprint(\"Starting Training on Full Dataset...\")\n\nhistory = model.fit(\n    train_gen,\n    validation_data=val_gen,\n    epochs=50,\n    callbacks=[checkpoint, early_stop],\n    verbose=1  # <--- CRITICAL: This enables the progress bar and metric logs\n)\n\n# Save the final state\nmodel.save('lipinc_full_data_final.h5')\nprint(\"Model saved.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T15:51:08.570317Z","iopub.execute_input":"2025-11-29T15:51:08.570507Z","iopub.status.idle":"2025-11-29T16:32:14.106776Z","shell.execute_reply.started":"2025-11-29T15:51:08.570491Z","shell.execute_reply":"2025-11-29T16:32:14.104474Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score, average_precision_score, jaccard_score, accuracy_score\n\nprint(\"\\nEvaluating on Test Set...\")\n# 1. Standard Keras Evaluate (Loss & Accuracy)\nresults = model.evaluate(test_gen, verbose=1)\nprint(f\"Test Loss: {results[0]:.4f}\")\nprint(f\"Test Accuracy: {results[-1]:.4f}\")\n\n# 2. Detailed Metrics Calculation\nprint(\"Calculating AP, AUC, and IoU...\")\ny_true_all = []\ny_pred_all = []\n\n# Reset generator to ensure we start from the beginning\ntest_gen.on_epoch_end()\n\n# Loop through the generator to gather all predictions\nfor i in range(len(test_gen)):\n    inputs, targets = test_gen[i]\n    \n    # Extract True Labels (One-hot encoded)\n    y_true_batch = targets['class_output']\n    \n    # Get Predictions\n    preds = model.predict_on_batch(inputs)\n    # preds is a list [class_output, features_output], we need class_output (index 0)\n    y_pred_batch = preds[0] \n    \n    y_true_all.extend(y_true_batch)\n    y_pred_all.extend(y_pred_batch)\n\n# Convert to numpy arrays\ny_true_all = np.array(y_true_all)\ny_pred_all = np.array(y_pred_all)\n\n# Extract probabilities for the \"Fake\" class (Index 1)\n# y_true_all is shape (N, 2) -> [Real, Fake]\n# y_pred_all is shape (N, 2) -> [Prob_Real, Prob_Fake]\ntrue_labels = y_true_all[:, 1]\npred_probs = y_pred_all[:, 1]\n\n# Convert probabilities to hard binary labels (0 or 1) for IoU calculation\n# Threshold is usually 0.5\npred_labels = (pred_probs > 0.5).astype(int)\n\ntry:\n    # 1. ROC-AUC\n    roc_auc = roc_auc_score(true_labels, pred_probs)\n    \n    # 2. Average Precision (AP)\n    ap_score = average_precision_score(true_labels, pred_probs)\n    \n    # 3. Intersection over Union (IoU) - Equivalent to Jaccard Score for binary classification\n    # Calculates: TP / (TP + FP + FN)\n    iou_score = jaccard_score(true_labels, pred_labels, average='binary')\n\n    print(\"-\" * 30)\n    print(f\"ROC-AUC  : {roc_auc:.4f}\")\n    print(f\"AP Score : {ap_score:.4f}\")\n    print(f\"IoU Score: {iou_score:.4f}\")\n    print(\"-\" * 30)\n    \nexcept Exception as e:\n    print(\"Error calculating metrics (Check if test set has both classes):\", e)","metadata":{"trusted":true,"execution":{"execution_failed":"2025-11-29T15:34:15.826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.metrics import confusion_matrix, classification_report\n\n# ==========================================\n# 1. Confusion Matrix & Classification Report\n# ==========================================\n\n# Ensure variables from the previous cell are available\n# true_labels: 1D array of actual class indices (0 or 1)\n# pred_labels: 1D array of predicted class indices (0 or 1)\n\n# Generate Confusion Matrix\ncm = confusion_matrix(true_labels, pred_labels)\n\nplt.figure(figsize=(6, 5))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues', \n            xticklabels=['Real', 'Fake'], yticklabels=['Real', 'Fake'])\nplt.xlabel('Predicted Label')\nplt.ylabel('True Label')\nplt.title('Confusion Matrix')\nplt.show()\n\n# Generate Classification Report\nprint(\"\\n--- Classification Report ---\")\nprint(classification_report(true_labels, pred_labels, target_names=['Real', 'Fake']))\n\n# ==========================================\n# 2. Training History Plots\n# ==========================================\n\n# We plot the specific loss and accuracy for the 'class_output' head\nacc = history.history['class_output_accuracy']\nval_acc = history.history['val_class_output_accuracy']\n\nloss = history.history['class_output_loss']\nval_loss = history.history['val_class_output_loss']\n\nepochs_range = range(len(acc))\n\nplt.figure(figsize=(14, 5))\n\n# Plot Accuracy\nplt.subplot(1, 2, 1)\nplt.plot(epochs_range, acc, label='Training Accuracy')\nplt.plot(epochs_range, val_acc, label='Validation Accuracy')\nplt.legend(loc='lower right')\nplt.title('Training and Validation Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\n\n# Plot Loss\nplt.subplot(1, 2, 2)\nplt.plot(epochs_range, loss, label='Training Loss')\nplt.plot(epochs_range, val_loss, label='Validation Loss')\nplt.legend(loc='upper right')\nplt.title('Training and Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}