{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"dockerImageVersionId":29844,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport shutil\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\n\n# Let's first explore what we actually have\nbase_path = \"/kaggle/input/deepfake-detection-challenge\"\ntrain_videos_path = f\"{base_path}/train_sample_videos\"\n\nprint(\"=== Exploring the dataset ===\")\nprint(\"Files in main directory:\")\nfor item in os.listdir(base_path):\n    print(f\"  {item}\")\n\nprint(f\"\\nFiles in train_sample_videos (first 10):\")\nif os.path.exists(train_videos_path):\n    train_files = os.listdir(train_videos_path)\n    for f in train_files[:10]:\n        print(f\"  {f}\")\n    print(f\"Total files in train_sample_videos: {len(train_files)}\")\nelse:\n    print(\"train_sample_videos directory not found!\")\n\n# Check if there's a metadata.json in train_sample_videos\nmetadata_path = f\"{train_videos_path}/metadata.json\"\nprint(f\"\\nChecking for metadata.json: {os.path.exists(metadata_path)}\")\n\n# If no metadata.json, let's check what's in sample_submission.csv\nsample_sub_path = f\"{base_path}/sample_submission.csv\"\nif os.path.exists(sample_sub_path):\n    print(f\"\\nReading sample_submission.csv:\")\n    df = pd.read_csv(sample_sub_path)\n    print(f\"Shape: {df.shape}\")\n    print(f\"Columns: {df.columns.tolist()}\")\n    print(f\"First 5 rows:\")\n    print(df.head())\n    \n    # Check if these files are in train_sample_videos\n    if os.path.exists(train_videos_path):\n        sample_files = df['filename'].tolist()[:5]\n        print(f\"\\nChecking if sample_submission files exist in train_sample_videos:\")\n        for f in sample_files:\n            exists = os.path.exists(f\"{train_videos_path}/{f}\")\n            print(f\"  {f}: {'✓' if exists else '✗'}\")\n\n# Try to find metadata.json in different locations\npossible_metadata_paths = [\n    f\"{train_videos_path}/metadata.json\",\n    f\"{base_path}/metadata.json\",\n    f\"{base_path}/train_metadata.json\"\n]\n\nmetadata_found = None\nfor path in possible_metadata_paths:\n    if os.path.exists(path):\n        metadata_found = path\n        print(f\"\\nFound metadata at: {path}\")\n        break\n\nif metadata_found:\n    print(\"Trying to read metadata...\")\n    try:\n        import json\n        with open(metadata_found, 'r') as f:\n            # Read first few characters to check format\n            content = f.read(200)\n            f.seek(0)  # Reset file pointer\n            print(f\"First 200 chars: {content}\")\n            \n            # Try to load JSON\n            metadata = json.load(f)\n            print(f\"Successfully loaded metadata with {len(metadata)} entries\")\n            \n            # Show sample\n            sample_key = next(iter(metadata))\n            print(f\"Sample entry: {sample_key} -> {metadata[sample_key]}\")\n            \n        # Now do the split\n        print(\"\\n=== Starting data split ===\")\n        \n        filenames = list(metadata.keys())\n        labels = [metadata[f]['label'] for f in filenames]\n        \n        print(f\"Total files: {len(filenames)}\")\n        print(f\"Label distribution: {pd.Series(labels).value_counts().to_dict()}\")\n        \n        # Split: 80-10-10\n        train_files, temp_files, train_labels, temp_labels = train_test_split(\n            filenames, labels, test_size=0.2, stratify=labels, random_state=42\n        )\n        \n        val_files, test_files = train_test_split(\n            temp_files, test_size=0.5, stratify=temp_labels, random_state=42\n        )\n        \n        print(f\"Train: {len(train_files)}, Val: {len(val_files)}, Test: {len(test_files)}\")\n        \n        # Create output directories\n        output_dir = \"/kaggle/working/split_data\"\n        for split in ['train', 'validation', 'test']:\n            os.makedirs(f\"{output_dir}/{split}\", exist_ok=True)\n        \n        # Copy files\n        splits = {'train': train_files, 'validation': val_files, 'test': test_files}\n        \n        for split_name, file_list in splits.items():\n            print(f\"\\nCopying {split_name} files...\")\n            copied = 0\n            for filename in file_list[:5]:  # Copy only first 5 for testing\n                src = f\"{train_videos_path}/{filename}\"\n                dst = f\"{output_dir}/{split_name}/{filename}\"\n                if os.path.exists(src):\n                    shutil.copy2(src, dst)\n                    copied += 1\n            print(f\"Copied {copied} files to {split_name}\")\n        \n        print(\"SUCCESS! Data split completed.\")\n        \n    except Exception as e:\n        print(f\"Error processing metadata: {e}\")\n        print(\"Let's try a different approach...\")\n        \nelse:\n    print(\"\\nNo metadata.json found. Let's work with what we have...\")\n    \n    # List all video files and create basic split\n    if os.path.exists(train_videos_path):\n        video_files = [f for f in os.listdir(train_videos_path) \n                      if f.endswith(('.mp4', '.avi', '.mov'))]\n        \n        print(f\"Found {len(video_files)} video files\")\n        \n        if len(video_files) > 0:\n            # Simple random split without labels\n            from sklearn.model_selection import train_test_split\n            \n            train_files, temp_files = train_test_split(video_files, test_size=0.2, random_state=42)\n            val_files, test_files = train_test_split(temp_files, test_size=0.5, random_state=42)\n            \n            print(f\"Split: Train={len(train_files)}, Val={len(val_files)}, Test={len(test_files)}\")\n            \n            # Create directories and copy files\n            output_dir = \"/kaggle/working/split_data\"\n            splits = {'train': train_files, 'validation': val_files, 'test': test_files}\n            \n            for split_name, file_list in splits.items():\n                os.makedirs(f\"{output_dir}/{split_name}\", exist_ok=True)\n                print(f\"Created {output_dir}/{split_name}\")\n            \n            print(\"Basic split completed without metadata!\")\n        else:\n            print(\"No video files found!\")\n    else:\n        print(\"train_sample_videos directory not found!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T11:29:05.428437Z","iopub.execute_input":"2025-07-30T11:29:05.428733Z","iopub.status.idle":"2025-07-30T11:29:09.679452Z","shell.execute_reply.started":"2025-07-30T11:29:05.428678Z","shell.execute_reply":"2025-07-30T11:29:09.678551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport os\nimport shutil\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\n\n# Paths\nbase_path = \"/kaggle/input/deepfake-detection-challenge\"\ntrain_videos_path = f\"{base_path}/train_sample_videos\"\nmetadata_path = f\"{train_videos_path}/metadata.json\"\noutput_dir = \"/kaggle/working/split_data\"\n\nprint(\"=== Loading metadata ===\")\nwith open(metadata_path, 'r') as f:\n    metadata = json.load(f)\n\nfilenames = list(metadata.keys())\nlabels = [metadata[f]['label'] for f in filenames]\n\nprint(f\"Total videos: {len(filenames)}\")\nprint(f\"REAL: {labels.count('REAL')}\")\nprint(f\"FAKE: {labels.count('FAKE')}\")\n\nprint(\"\\n=== Splitting data (80-10-10) ===\")\n# Split: 80% train, 10% validation, 10% test\ntrain_files, temp_files, train_labels, temp_labels = train_test_split(\n    filenames, labels, test_size=0.2, stratify=labels, random_state=42\n)\n\nval_files, test_files = train_test_split(\n    temp_files, test_size=0.5, stratify=temp_labels, random_state=42\n)\n\nprint(f\"Train: {len(train_files)} videos\")\nprint(f\"Validation: {len(val_files)} videos\") \nprint(f\"Test: {len(test_files)} videos\")\n\n# Verify label distribution in each split\ndef count_labels(file_list):\n    labels = [metadata[f]['label'] for f in file_list]\n    return {'REAL': labels.count('REAL'), 'FAKE': labels.count('FAKE')}\n\nprint(f\"\\nLabel distribution:\")\nprint(f\"Train: {count_labels(train_files)}\")\nprint(f\"Val: {count_labels(val_files)}\")\nprint(f\"Test: {count_labels(test_files)}\")\n\nprint(\"\\n=== Creating directories ===\")\nfor split in ['train', 'validation', 'test']:\n    os.makedirs(f\"{output_dir}/{split}\", exist_ok=True)\n    print(f\"Created {output_dir}/{split}/\")\n\nprint(\"\\n=== Copying all files ===\")\nsplits = {\n    'train': train_files,\n    'validation': val_files, \n    'test': test_files\n}\n\ntotal_copied = 0\nfor split_name, file_list in splits.items():\n    print(f\"\\nCopying {split_name} files...\")\n    copied = 0\n    missing = 0\n    \n    for i, filename in enumerate(file_list):\n        src = f\"{train_videos_path}/{filename}\"\n        dst = f\"{output_dir}/{split_name}/{filename}\"\n        \n        if os.path.exists(src):\n            shutil.copy2(src, dst)\n            copied += 1\n        else:\n            missing += 1\n        \n        # Progress indicator\n        if (i + 1) % 50 == 0 or i == len(file_list) - 1:\n            print(f\"  Progress: {i + 1}/{len(file_list)} - Copied: {copied}, Missing: {missing}\")\n    \n    total_copied += copied\n    print(f\"  ✓ {split_name}: {copied} files copied\")\n\nprint(f\"\\n=== Creating metadata files ===\")\n# Create metadata for each split\ntrain_metadata = {f: metadata[f] for f in train_files}\nval_metadata = {f: metadata[f] for f in val_files}\ntest_metadata = {f: metadata[f] for f in test_files}\n\n# Save metadata files\nwith open(f\"{output_dir}/train_metadata.json\", 'w') as f:\n    json.dump(train_metadata, f, indent=2)\n\nwith open(f\"{output_dir}/val_metadata.json\", 'w') as f:\n    json.dump(val_metadata, f, indent=2)\n\nwith open(f\"{output_dir}/test_metadata.json\", 'w') as f:\n    json.dump(test_metadata, f, indent=2)\n\nprint(f\"✓ Saved train_metadata.json ({len(train_metadata)} entries)\")\nprint(f\"✓ Saved val_metadata.json ({len(val_metadata)} entries)\")\nprint(f\"✓ Saved test_metadata.json ({len(test_metadata)} entries)\")\n\nprint(\"\\n=== Final Summary ===\")\nprint(f\"Total files copied: {total_copied}\")\nprint(f\"Data saved to: {output_dir}\")\nprint(f\"\\nDirectory structure:\")\nprint(f\"{output_dir}/\")\nprint(f\"├── train/ ({len(train_files)} videos)\")\nprint(f\"├── validation/ ({len(val_files)} videos)\")\nprint(f\"├── test/ ({len(test_files)} videos)\")\nprint(f\"├── train_metadata.json\")\nprint(f\"├── val_metadata.json\")\nprint(f\"└── test_metadata.json\")\n\nprint(f\"\\n🎉 Split completed successfully!\")\nprint(f\"You can now use these paths for training:\")\nprint(f\"TRAIN_DIR = '{output_dir}/train'\")\nprint(f\"VAL_DIR = '{output_dir}/validation'\")\nprint(f\"TEST_DIR = '{output_dir}/test'\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T11:29:20.252988Z","iopub.execute_input":"2025-07-30T11:29:20.253335Z","iopub.status.idle":"2025-07-30T11:29:34.581728Z","shell.execute_reply.started":"2025-07-30T11:29:20.253282Z","shell.execute_reply":"2025-07-30T11:29:34.580224Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport numpy as np\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout, GlobalAveragePooling2D\nfrom tensorflow.keras.applications import Xception, InceptionV3, MobileNet\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import EarlyStopping\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score, precision_score, recall_score, f1_score\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom collections import Counter\nimport json\nimport cv2\nfrom sklearn.utils.class_weight import compute_class_weight\n\n# UPDATED DATA PATHS - Using split data\ntrain_data_dir = '/kaggle/working/split_data/train'\nvalid_data_dir = '/kaggle/working/split_data/validation'  \ntest_data_dir = '/kaggle/working/split_data/test'\n\n# Load metadata for class weights calculation\nwith open('/kaggle/working/split_data/train_metadata.json', 'r') as f:\n    train_metadata = json.load(f)\nwith open('/kaggle/working/split_data/val_metadata.json', 'r') as f:\n    val_metadata = json.load(f)\nwith open('/kaggle/working/split_data/test_metadata.json', 'r') as f:\n    test_metadata = json.load(f)\n\nprint(f\"Train samples: {len(train_metadata)}\")\nprint(f\"Validation samples: {len(val_metadata)}\")\nprint(f\"Test samples: {len(test_metadata)}\")\n\n# Image preprocessing settings\nimg_width, img_height = 224, 224\nbatch_size = 16\nframes_per_video = 3\n\ndef extract_frames_from_video(video_path, num_frames=3):\n    \"\"\"Extract frames from video efficiently\"\"\"\n    frames = []\n    cap = cv2.VideoCapture(video_path)\n    \n    if not cap.isOpened():\n        return None\n    \n    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    if total_frames == 0:\n        cap.release()\n        return None\n    \n    # Select frame indices evenly distributed\n    frame_indices = np.linspace(0, total_frames - 1, num_frames, dtype=int)\n    \n    for frame_idx in frame_indices:\n        cap.set(cv2.CAP_PROP_POS_FRAMES, frame_idx)\n        ret, frame = cap.read()\n        \n        if ret:\n            # Resize and normalize\n            frame = cv2.resize(frame, (img_width, img_height))\n            frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n            frame = frame.astype(np.float32) / 255.0\n            frames.append(frame)\n    \n    cap.release()\n    \n    # If we couldn't extract enough frames, pad with copies of the last frame\n    while len(frames) < num_frames:\n        if frames:\n            frames.append(frames[-1])\n        else:\n            return None\n    \n    return np.array(frames)\n\nclass VideoDataGenerator:\n    \"\"\"Custom data generator for videos\"\"\"\n    \n    def __init__(self, metadata, video_dir, batch_size=16, frames_per_video=3):\n        self.metadata = metadata\n        self.video_dir = video_dir\n        self.batch_size = batch_size\n        self.frames_per_video = frames_per_video\n        self.video_files = list(metadata.keys())\n        self.indices = np.arange(len(self.video_files))\n        self.shuffle()\n    \n    def shuffle(self):\n        np.random.shuffle(self.indices)\n    \n    def __len__(self):\n        return len(self.video_files) // self.batch_size\n    \n    def __getitem__(self, idx):\n        batch_indices = self.indices[idx * self.batch_size:(idx + 1) * self.batch_size]\n        \n        X = []\n        y = []\n        \n        for i in batch_indices:\n            video_file = self.video_files[i]\n            video_path = os.path.join(self.video_dir, video_file)\n            \n            # Extract frames\n            frames = extract_frames_from_video(video_path, self.frames_per_video)\n            \n            if frames is not None:\n                # Average frames to get single image\n                avg_frame = np.mean(frames, axis=0)\n                X.append(avg_frame)\n                \n                # Get label (FAKE=0, REAL=1)\n                label = 1 if self.metadata[video_file]['label'] == 'REAL' else 0\n                y.append(label)\n        \n        if len(X) == 0:\n            # Return dummy data if no valid frames\n            X = [np.zeros((img_height, img_width, 3))]\n            y = [0]\n        \n        return np.array(X), np.array(y)\n\n# Create data generators\nprint(\"Creating data generators...\")\ntrain_generator = VideoDataGenerator(train_metadata, train_data_dir, batch_size=batch_size)\nvalidation_generator = VideoDataGenerator(val_metadata, valid_data_dir, batch_size=batch_size)\ntest_generator = VideoDataGenerator(test_metadata, test_data_dir, batch_size=batch_size)\n\nprint(f\"Train batches: {len(train_generator)}\")\nprint(f\"Validation batches: {len(validation_generator)}\")\nprint(f\"Test batches: {len(test_generator)}\")\n\n# Calculate class weights for imbalanced dataset\ndef calculate_class_weights():\n    \"\"\"Calculate class weights for imbalanced dataset\"\"\"\n    # Count samples in each class\n    train_labels = [info['label'] for info in train_metadata.values()]\n    real_count = train_labels.count('REAL')\n    fake_count = train_labels.count('FAKE')\n    \n    total = real_count + fake_count\n    \n    # Calculate weights (inverse frequency)\n    weight_real = total / (2 * real_count)\n    weight_fake = total / (2 * fake_count)\n    \n    class_weights = {0: weight_fake, 1: weight_real}  # 0=FAKE, 1=REAL\n    \n    print(f\"Class distribution - REAL: {real_count}, FAKE: {fake_count}\")\n    print(f\"Class weights - FAKE: {weight_fake:.3f}, REAL: {weight_real:.3f}\")\n    \n    return class_weights\n\nclass_weights = calculate_class_weights()\n\n# Test data generator and visualize\nprint(\"\\nTesting data generator...\")\ntry:\n    X_batch, y_batch = train_generator[0]\n    print(f\"Batch shape: {X_batch.shape}\")\n    print(f\"Labels shape: {y_batch.shape}\")\n    print(f\"Sample labels: {y_batch[:5]}\")\n    \n    # Visualize some samples\n    plt.figure(figsize=(15, 3))\n    for i in range(min(5, len(X_batch))):\n        plt.subplot(1, 5, i+1)\n        plt.imshow(X_batch[i])\n        label_name = 'REAL' if y_batch[i] == 1 else 'FAKE'\n        plt.title(f'{label_name}')\n        plt.axis('off')\n    plt.suptitle(\"Sample Training Images (Extracted from Videos)\")\n    plt.show()\n    \nexcept Exception as e:\n    print(f\"Error in data generator: {e}\")\n\n# Calculate steps per epoch\ntrain_steps_per_epoch = max(1, len(train_generator))\nvalid_steps_per_epoch = max(1, len(validation_generator))\n\nprint(f\"Training steps per epoch: {train_steps_per_epoch}\")\nprint(f\"Validation steps per epoch: {valid_steps_per_epoch}\")\n\n# Training function\ndef train_model_with_generator(model, model_name, train_gen, val_gen, epochs=5):\n    \"\"\"Train model with custom generator\"\"\"\n    print(f\"\\n{'='*50}\")\n    print(f\"Training {model_name} Model\")\n    print(f\"{'='*50}\")\n    \n    history = {'accuracy': [], 'val_accuracy': [], 'loss': [], 'val_loss': []}\n    \n    for epoch in range(epochs):\n        print(f\"\\nEpoch {epoch+1}/{epochs}\")\n        \n        # Training\n        train_loss = 0\n        train_acc = 0\n        train_batches = 0\n        \n        train_gen.shuffle()\n        \n        for batch_idx in range(len(train_gen)):\n            X_batch, y_batch = train_gen[batch_idx]\n            \n            if len(X_batch) == 0:\n                continue\n                \n            # Train on batch\n            batch_history = model.train_on_batch(X_batch, y_batch, class_weight=class_weights)\n            train_loss += batch_history[0]\n            train_acc += batch_history[1]\n            train_batches += 1\n            \n            if batch_idx % 5 == 0:\n                print(f\"  Batch {batch_idx+1}/{len(train_gen)} - Loss: {batch_history[0]:.4f}, Acc: {batch_history[1]:.4f}\")\n        \n        # Validation\n        val_loss = 0\n        val_acc = 0\n        val_batches = 0\n        \n        for batch_idx in range(len(val_gen)):\n            X_batch, y_batch = val_gen[batch_idx]\n            \n            if len(X_batch) == 0:\n                continue\n                \n            batch_history = model.test_on_batch(X_batch, y_batch)\n            val_loss += batch_history[0]\n            val_acc += batch_history[1]\n            val_batches += 1\n        \n        # Calculate averages\n        if train_batches > 0:\n            train_loss /= train_batches\n            train_acc /= train_batches\n        \n        if val_batches > 0:\n            val_loss /= val_batches\n            val_acc /= val_batches\n        \n        # Store history\n        history['loss'].append(train_loss)\n        history['accuracy'].append(train_acc)\n        history['val_loss'].append(val_loss)\n        history['val_accuracy'].append(val_acc)\n        \n        print(f\"  Train - Loss: {train_loss:.4f}, Acc: {train_acc:.4f}\")\n        print(f\"  Val   - Loss: {val_loss:.4f}, Acc: {val_acc:.4f}\")\n    \n    return history\n\n# 1. CNN Model\ndef create_cnn_model():\n    \"\"\"Create a simple CNN model\"\"\"\n    model = Sequential([\n        Conv2D(32, (3,3), activation='relu', input_shape=(img_height, img_width, 3)),\n        MaxPooling2D(2, 2),\n        Conv2D(64, (3,3), activation='relu'),\n        MaxPooling2D(2, 2),\n        Conv2D(128, (3,3), activation='relu'),\n        MaxPooling2D(2, 2),\n        Conv2D(256, (3,3), activation='relu'),\n        MaxPooling2D(2, 2),\n        Flatten(),\n        Dense(512, activation='relu'),\n        Dropout(0.5),\n        Dense(256, activation='relu'),\n        Dropout(0.5),\n        Dense(1, activation='sigmoid')\n    ])\n    return model\n\n# 2. Xception Model\ndef create_xception_model():\n    \"\"\"Create Xception-based model\"\"\"\n    base_model = Xception(weights='imagenet', include_top=False, input_shape=(img_height, img_width, 3))\n    \n    # Freeze base model initially\n    base_model.trainable = False\n    \n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(256, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    x = Dense(128, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    predictions = Dense(1, activation='sigmoid')(x)\n    \n    model = Model(inputs=base_model.input, outputs=predictions)\n    return model\n\n# 3. InceptionV3 Model\ndef create_inception_model():\n    \"\"\"Create InceptionV3-based model\"\"\"\n    base_model = InceptionV3(weights='imagenet', include_top=False, input_shape=(img_height, img_width, 3))\n    \n    # Freeze base model initially\n    base_model.trainable = False\n    \n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(256, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    x = Dense(128, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    predictions = Dense(1, activation='sigmoid')(x)\n    \n    model = Model(inputs=base_model.input, outputs=predictions)\n    return model\n\n# 4. MobileNet Model\ndef create_mobilenet_model():\n    \"\"\"Create MobileNet-based model\"\"\"\n    base_model = MobileNet(weights='imagenet', include_top=False, input_shape=(img_height, img_width, 3))\n    \n    # Freeze base model initially\n    base_model.trainable = False\n    \n    x = base_model.output\n    x = GlobalAveragePooling2D()(x)\n    x = Dense(256, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    x = Dense(128, activation='relu')(x)\n    x = Dropout(0.5)(x)\n    predictions = Dense(1, activation='sigmoid')(x)\n    \n    model = Model(inputs=base_model.input, outputs=predictions)\n    return model\n\n# Create and compile all models\nprint(\"\\nCreating and compiling models...\")\n\nmodel_cnn = create_cnn_model()\nmodel_cnn.compile(loss='binary_crossentropy', optimizer=Adam(learning_rate=0.001), metrics=['accuracy'])\n\nxception_model = create_xception_model()\nxception_model.compile(optimizer=Adam(learning_rate=0.0001), loss='binary_crossentropy', metrics=['accuracy'])\n\ninception_model = create_inception_model()\ninception_model.compile(optimizer=Adam(learning_rate=0.0001), loss='binary_crossentropy', metrics=['accuracy'])\n\nmobilenet_model = create_mobilenet_model()\nmobilenet_model.compile(optimizer=Adam(learning_rate=0.0001), loss='binary_crossentropy', metrics=['accuracy'])\n\nprint(\"All models created and compiled successfully!\")\n\n# Train all models\nmodels = [\n    (model_cnn, \"CNN\", 5),\n    (xception_model, \"Xception\", 3),\n    (inception_model, \"InceptionV3\", 3),  \n    (mobilenet_model, \"MobileNet\", 3)\n]\n\nhistories = {}\ntrained_models = {}\n\nfor model, name, epochs in models:\n    print(f\"\\n{name} Model Summary:\")\n    if name == \"CNN\":\n        model.summary()\n    \n    history = train_model_with_generator(model, name, train_generator, validation_generator, epochs=epochs)\n    histories[name] = history\n    trained_models[name] = model\n    \n    # Plot training history\n    plt.figure(figsize=(12, 6))\n    \n    plt.subplot(1, 2, 1)\n    plt.plot(history['accuracy'], 'b-', label='Train Accuracy')\n    plt.plot(history['val_accuracy'], 'r-', label='Val Accuracy')\n    plt.title(f'{name} Model Accuracy')\n    plt.xlabel('Epoch')\n    plt.ylabel('Accuracy')\n    plt.legend()\n    \n    plt.subplot(1, 2, 2)\n    plt.plot(history['loss'], 'b-', label='Train Loss')\n    plt.plot(history['val_loss'], 'r-', label='Val Loss')\n    plt.title(f'{name} Model Loss')\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')\n    plt.legend()\n    \n    plt.tight_layout()\n    plt.show()\n\n# Evaluation Function\ndef evaluate_model(model, model_name, test_gen):\n    \"\"\"Evaluate a single model and display results\"\"\"\n    print(f\"\\nEvaluating {model_name} Model...\")\n    \n    # Get predictions\n    predictions = []\n    labels = []\n    \n    for batch_idx in range(len(test_gen)):\n        X_batch, y_batch = test_gen[batch_idx]\n        \n        if len(X_batch) == 0:\n            continue\n        \n        batch_pred = model.predict(X_batch, verbose=0)\n        predictions.extend(batch_pred.flatten())\n        labels.extend(y_batch)\n    \n    # Convert to binary predictions\n    y_pred_binary = (np.array(predictions) > 0.5).astype(int)\n    y_true = np.array(labels)\n    \n    # Calculate metrics\n    accuracy = accuracy_score(y_true, y_pred_binary)\n    precision = precision_score(y_true, y_pred_binary)\n    recall = recall_score(y_true, y_pred_binary)\n    f1 = f1_score(y_true, y_pred_binary)\n    \n    print(f\"{model_name} Results:\")\n    print(f\"  Accuracy:  {accuracy:.4f}\")\n    print(f\"  Precision: {precision:.4f}\")\n    print(f\"  Recall:    {recall:.4f}\")\n    print(f\"  F1-Score:  {f1:.4f}\")\n    \n    # Classification report\n    print(f\"\\n{model_name} Classification Report:\")\n    print(classification_report(y_true, y_pred_binary, target_names=['FAKE', 'REAL']))\n    \n    # Confusion matrix\n    conf_matrix = confusion_matrix(y_true, y_pred_binary)\n    plt.figure(figsize=(7, 5))\n    sns.heatmap(conf_matrix, annot=True, cmap='Blues', fmt='g', \n                xticklabels=['FAKE', 'REAL'], yticklabels=['FAKE', 'REAL'])\n    plt.title(f'{model_name} Confusion Matrix')\n    plt.xlabel('Predicted labels')\n    plt.ylabel('True labels')\n    plt.show()\n    \n    return y_pred_binary, accuracy\n\n# Evaluate all models\nprint(\"=\"*50)\nprint(\"MODEL EVALUATION\")\nprint(\"=\"*50)\n\nresults = {}\nfor name, model in trained_models.items():\n    pred, acc = evaluate_model(model, name, test_generator)\n    results[name] = {'predictions': pred, 'accuracy': acc}\n\n# Ensemble Method\ndef ensemble_predict(models_dict, test_gen):\n    \"\"\"Create ensemble predictions using majority voting\"\"\"\n    print(f\"\\nCreating Ensemble Prediction...\")\n    \n    all_predictions = []\n    \n    # Get predictions from each model (skip CNN for ensemble)\n    for name, model in models_dict.items():\n        if name == \"CNN\":  # Skip CNN for ensemble\n            continue\n            \n        predictions = []\n        for batch_idx in range(len(test_gen)):\n            X_batch, y_batch = test_gen[batch_idx]\n            if len(X_batch) == 0:\n                continue\n            batch_pred = model.predict(X_batch, verbose=0)\n            predictions.extend(batch_pred.flatten())\n        \n        pred_binary = (np.array(predictions) > 0.5).astype(int)\n        all_predictions.append(pred_binary)\n        print(f\"  {name} predictions added\")\n    \n    # Majority voting\n    all_predictions = np.array(all_predictions)\n    ensemble_pred = np.apply_along_axis(\n        lambda x: Counter(x).most_common(1)[0][0], \n        axis=0, \n        arr=all_predictions\n    )\n    \n    return ensemble_pred\n\n# Create ensemble\nprint(\"\\n\" + \"=\"*50)\nprint(\"ENSEMBLE MODEL EVALUATION\")\nprint(\"=\"*50)\n\nensemble_pred = ensemble_predict(trained_models, test_generator)\n\n# Get test labels\ntest_labels = []\nfor batch_idx in range(len(test_generator)):\n    X_batch, y_batch = test_generator[batch_idx]\n    if len(X_batch) == 0:\n        continue\n    test_labels.extend(y_batch)\n\ntest_labels = np.array(test_labels)\n\n# Ensemble evaluation\nensemble_acc = accuracy_score(test_labels, ensemble_pred)\nensemble_precision = precision_score(test_labels, ensemble_pred)\nensemble_recall = recall_score(test_labels, ensemble_pred)\nensemble_f1 = f1_score(test_labels, ensemble_pred)\n\nprint(f\"Ensemble Results:\")\nprint(f\"  Accuracy:  {ensemble_acc:.4f}\")\nprint(f\"  Precision: {ensemble_precision:.4f}\")  \nprint(f\"  Recall:    {ensemble_recall:.4f}\")\nprint(f\"  F1-Score:  {ensemble_f1:.4f}\")\n\nprint(\"\\nEnsemble Classification Report:\")\nprint(classification_report(test_labels, ensemble_pred, target_names=['FAKE', 'REAL']))\n\n# Ensemble confusion matrix\nconf_matrix_ensemble = confusion_matrix(test_labels, ensemble_pred)\nplt.figure(figsize=(7, 5))\nsns.heatmap(conf_matrix_ensemble, annot=True, cmap='Blues', fmt='g', \n            xticklabels=['FAKE', 'REAL'], yticklabels=['FAKE', 'REAL'])\nplt.title('Ensemble Confusion Matrix')\nplt.xlabel('Predicted labels')\nplt.ylabel('True labels')\nplt.show()\n\n# Summary of all model performances\nprint(\"\\n\" + \"=\"*50)\nprint(\"SUMMARY OF MODEL PERFORMANCES\")\nprint(\"=\"*50)\n\nall_accuracies = [(name, results[name]['accuracy']) for name in results.keys()]\nall_accuracies.append((\"Ensemble\", ensemble_acc))\n\nfor name, acc in sorted(all_accuracies, key=lambda x: x[1], reverse=True):\n    print(f\"{name:12} Accuracy: {acc:.4f}\")\n\n# Find best performing model\nbest_model_name = max(all_accuracies, key=lambda x: x[1])[0]\nprint(f\"\\nBest performing model: {best_model_name}\")\n\n# Save all models\nprint(f\"\\nSaving models...\")\nmodel_cnn.save('/kaggle/working/cnn_model.h5')\nxception_model.save('/kaggle/working/xception_model.h5')\ninception_model.save('/kaggle/working/inception_model.h5')\nmobilenet_model.save('/kaggle/working/mobilenet_model.h5')\n\nprint(f\"All models saved:\")\nprint(f\"  CNN: /kaggle/working/cnn_model.h5\")\nprint(f\"  Xception: /kaggle/working/xception_model.h5\")\nprint(f\"  InceptionV3: /kaggle/working/inception_model.h5\")\nprint(f\"  MobileNet: /kaggle/working/mobilenet_model.h5\")\n\n# Save best individual model\nif best_model_name != \"Ensemble\":\n    best_model = trained_models[best_model_name]\n    best_model.save('/kaggle/working/best_deepfake_model.h5')\n    print(f\"Best individual model saved as: /kaggle/working/best_deepfake_model.h5\")\n\nprint(f\"\\nTraining completed successfully!\")\nprint(f\"All 4 models + ensemble trained and evaluated!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-30T11:45:22.772810Z","iopub.execute_input":"2025-07-30T11:45:22.773313Z","iopub.status.idle":"2025-07-30T11:47:08.632902Z","shell.execute_reply.started":"2025-07-30T11:45:22.773231Z","shell.execute_reply":"2025-07-30T11:47:08.631105Z"}},"outputs":[],"execution_count":null}]}