{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":16880,"databundleVersionId":858837}],"dockerImageVersionId":31329,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:15:35.169379Z","iopub.execute_input":"2026-03-31T16:15:35.170053Z","iopub.status.idle":"2026-03-31T16:15:39.420548Z","shell.execute_reply.started":"2026-03-31T16:15:35.170016Z","shell.execute_reply":"2026-03-31T16:15:39.419723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import json\nimport pandas as pd\nimport glob\n\n# Mencari file metadata.json secara otomatis di seluruh folder input\n# agar tidak terjadi FileNotFoundError lagi\nmetadata_files = glob.glob('/kaggle/input/**/metadata.json', recursive=True)\n\nif len(metadata_files) > 0:\n    meta_path = metadata_files[0] # Ambil file metadata pertama yang ditemukan\n    print(f\"File ketemu di: {meta_path}\")\n    \n    with open(meta_path) as f:\n        data = json.load(f)\n\n    # Ubah ke tabel\n    df_meta = pd.DataFrame(data).T\n    df_meta.index.name = 'filename'\n    df_meta = df_meta.reset_index()\n\n    print(f\"Total video terdeteksi: {len(df_meta)}\")\n    display(df_meta.head(10))\nelse:\n    print(\"Waduh, file metadata.json belum ketemu. Coba cek lagi apakah dataset sudah ter-ekstrak sempurna di panel Input.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:15:39.421959Z","iopub.execute_input":"2026-03-31T16:15:39.422367Z","iopub.status.idle":"2026-03-31T16:15:39.777535Z","shell.execute_reply.started":"2026-03-31T16:15:39.422335Z","shell.execute_reply":"2026-03-31T16:15:39.776822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\nimport os\nimport glob\n\n# 1. Ambil nama file dari tabel metadata\nvideo_file = df_meta.iloc[0]['filename']\n\n# 2. Gunakan glob untuk mencari lokasi asli file tersebut di seluruh folder input\n# Ini cara paling aman agar tidak kena FileNotFoundError lagi\nsearch_path = glob.glob(f'/kaggle/input/**/{video_file}', recursive=True)\n\nif len(search_path) > 0:\n    actual_video_path = search_path[0]\n    print(f\"Video ditemukan di: {actual_video_path}\")\n    \n    # 3. Mulai proses pembacaan video\n    cap = cv2.VideoCapture(actual_video_path)\n    ret, frame = cap.read()\n\n    if ret:\n        frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        plt.figure(figsize=(10, 6))\n        plt.imshow(frame)\n        plt.title(f\"Visualisasi: {video_file} ({df_meta.iloc[0]['label']})\")\n        plt.axis('off')\n        plt.show()\n    else:\n        print(\"Gagal membaca frame video.\")\n    cap.release()\nelse:\n    print(f\"Waduh, file {video_file} benar-benar tidak ditemukan di folder input!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:15:39.778720Z","iopub.execute_input":"2026-03-31T16:15:39.779159Z","iopub.status.idle":"2026-03-31T16:15:41.618475Z","shell.execute_reply.started":"2026-03-31T16:15:39.779121Z","shell.execute_reply":"2026-03-31T16:15:41.617557Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\n# 1. Load detektor wajah bawaan OpenCV (sudah ada di sistem, tidak perlu download)\nface_cascade = cv2.CascadeClassifier(cv2.data.haarcascades + 'haarcascade_frontalface_default.xml')\n\n# 2. Gunakan frame yang tadi sudah berhasil kita munculkan\n# (Pastikan cell yang menampilkan pria pirang tadi sudah pernah dijalankan)\ngray = cv2.cvtColor(frame, cv2.COLOR_RGB2GRAY)\n\n# 3. Cari wajah\nfaces = face_cascade.detectMultiScale(gray, 1.1, 4)\n\nif len(faces) > 0:\n    for (x, y, w, h) in faces:\n        # Potong wajah\n        face_crop = frame[y:y+h, x:x+w]\n        \n        # Tampilkan\n        plt.figure(figsize=(5, 5))\n        plt.imshow(face_crop)\n        plt.title(\"Wajah Berhasil Dipotong (Pakai OpenCV)\")\n        plt.axis('off')\n        plt.show()\n    print(f\"✅ Berhasil! Ditemukan {len(faces)} wajah.\")\nelse:\n    print(\"❌ Wajah tidak terdeteksi, coba ganti parameter detectMultiScale.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:15:41.621026Z","iopub.execute_input":"2026-03-31T16:15:41.621508Z","iopub.status.idle":"2026-03-31T16:15:43.064593Z","shell.execute_reply.started":"2026-03-31T16:15:41.621470Z","shell.execute_reply":"2026-03-31T16:15:43.063789Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# 1. Buat folder untuk menampung hasil crop\nos.makedirs('/kaggle/working/train_faces', exist_ok=True)\n\ndef extract_faces_from_videos(metadata_df, num_videos=10, frames_per_video=5):\n    face_count = 0\n    \n    for i, row in metadata_df.head(num_videos).iterrows():\n        video_name = row['filename']\n        label = row['label']\n        \n        # Cari path video secara otomatis\n        video_path = glob.glob(f'/kaggle/input/**/{video_name}', recursive=True)[0]\n        cap = cv2.VideoCapture(video_path)\n        \n        count = 0\n        while count < frames_per_video:\n            ret, frame = cap.read()\n            if not ret: break\n            \n            # Deteksi wajah pakai OpenCV (yang tadi sudah berhasil)\n            gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)\n            faces = face_cascade.detectMultiScale(gray, 1.1, 4)\n            \n            for (x, y, w, h) in faces:\n                face_crop = frame[y:y+h, x:x+w]\n                face_crop = cv2.resize(face_crop, (224, 224)) # Standarisasi ukuran untuk CNN\n                \n                # Simpan file: label_namafile_frameke.jpg\n                save_path = f'/kaggle/working/train_faces/{label}_{video_name}_{count}.jpg'\n                cv2.imwrite(save_path, cv2.cvtColor(face_crop, cv2.COLOR_RGB2BGR))\n                face_count += 1\n                break # Cukup ambil satu wajah per frame\n            \n            count += 1\n        cap.release()\n        print(f\"Selesai memproses: {video_name}\")\n\n    print(f\"✅ Total wajah yang berhasil dikumpulkan: {face_count}\")\n\n# Jalankan untuk 20 video pertama sebagai tes\nextract_faces_from_videos(df_meta, num_videos=20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:15:43.065507Z","iopub.execute_input":"2026-03-31T16:15:43.065844Z","iopub.status.idle":"2026-03-31T16:16:23.628779Z","shell.execute_reply.started":"2026-03-31T16:15:43.065808Z","shell.execute_reply":"2026-03-31T16:16:23.628230Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# 1. Pastikan library dasar tersedia dengan versi yang pas\n!pip install torchvision --no-deps -q\n\nimport torch\nimport torch.nn as nn\ntry:\n    import torchvision.models as models\n    print(\"✅ Berhasil memuat Torchvision!\")\nexcept ImportError:\n    print(\"❌ Masih ada kendala, mencoba cara alternatif...\")\n    # Jika cara di atas gagal, kita panggil langsung dari torch\n    from torch.hub import load_state_dict_from_url\n\n# 2. Definisikan ulang arsitekturnya (Sama seperti sebelumnya)\nclass CNNLSTM(nn.Module):\n    def __init__(self, num_classes=2):\n        super(CNNLSTM, self).__init__()\n        # Menggunakan ResNet18 sebagai ekstraktor fitur\n        resnet = models.resnet18(weights='IMAGENET1K_V1')\n        self.feature_extractor = nn.Sequential(*list(resnet.children())[:-1])\n        \n        self.lstm = nn.LSTM(input_size=512, hidden_size=256, num_layers=1, batch_first=True)\n        self.fc = nn.Linear(256, num_classes)\n        \n    def forward(self, x):\n        batch_size, seq_len, c, h, w = x.shape\n        # Meratakan dimensi untuk CNN\n        ii = x.view(batch_size * seq_len, c, h, w)\n        features = self.feature_extractor(ii)\n        features = features.view(batch_size, seq_len, -1)\n        \n        # Masukkan ke LSTM\n        lstm_out, _ = self.lstm(features)\n        last_time_step = lstm_out[:, -1, :]\n        \n        return self.fc(last_time_step)\n\n# 3. Inisialisasi Model\ndevice = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nmodel = CNNLSTM().to(device)\nprint(\"🚀 Model CNN-LSTM sudah siap di memori!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:16:23.629409Z","iopub.execute_input":"2026-03-31T16:16:23.629627Z","iopub.status.idle":"2026-03-31T16:16:25.155362Z","shell.execute_reply.started":"2026-03-31T16:16:23.629604Z","shell.execute_reply":"2026-03-31T16:16:25.154608Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch.optim as optim\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import transforms\nfrom PIL import Image\n\n# 1. Persiapan Data (Transformasi gambar agar seragam)\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n])\n\n# 2. Fungsi untuk mengambil data dari folder\nclass DeepfakeDataset(Dataset):\n    def __init__(self, folder, transform=None):\n        # Pastikan folder ada sebelum dibaca\n        if not os.path.exists(folder):\n            os.makedirs(folder)\n        self.files = [os.path.join(folder, f) for f in os.listdir(folder) if f.endswith('.jpg')]\n        self.transform = transform\n    def __len__(self): return len(self.files)\n    def __getitem__(self, idx):\n        img_path = self.files[idx]\n        image = Image.open(img_path).convert('RGB')\n        label = 1 if 'FAKE' in img_path else 0\n        if self.transform: image = self.transform(image)\n        # Menyesuaikan input untuk LSTM: (seq_len=1, channels, h, w)\n        return image.unsqueeze(0), label\n\n# 3. Setup Training\n# Kita pakai folder 'train_faces' hasil preprocessing tadi\ndataset = DeepfakeDataset('/kaggle/working/train_faces', transform=transform)\n\nif len(dataset) > 0:\n    train_loader = DataLoader(dataset, batch_size=4, shuffle=True)\n    criterion = torch.nn.CrossEntropyLoss()\n    optimizer = optim.Adam(model.parameters(), lr=0.001)\n\n    # 4. Mulai Training\n    print(f\"Memulai Training dengan {len(dataset)} gambar...\")\n    model.train()\n    for epoch in range(5):\n        running_loss = 0.0\n        for images, labels in train_loader:\n            images, labels = images.to(device), labels.to(device)\n            optimizer.zero_grad()\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            running_loss += loss.item()\n        print(f\"Epoch {epoch+1} - Loss: {running_loss/len(train_loader):.4f}\")\n    print(\"✅ Training Selesai!\")\nelse:\n    print(\"❌ Folder wajah kosong! Silakan jalankan ulang cell ekstraksi wajah tadi.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:23:12.534011Z","iopub.execute_input":"2026-03-31T16:23:12.534412Z","iopub.status.idle":"2026-03-31T16:23:15.432096Z","shell.execute_reply.started":"2026-03-31T16:23:12.534367Z","shell.execute_reply":"2026-03-31T16:23:15.431469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import Video\n\n# Ambil path video pertama yang kita temukan tadi\nvideo_path_display = glob.glob('/kaggle/input/**/aagfhgtpmv.mp4', recursive=True)[0]\n\n# Tampilkan video di dalam notebook\nVideo(video_path_display, embed=True, width=600)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:27:54.200981Z","iopub.execute_input":"2026-03-31T16:27:54.201617Z","iopub.status.idle":"2026-03-31T16:27:56.692740Z","shell.execute_reply.started":"2026-03-31T16:27:54.201586Z","shell.execute_reply":"2026-03-31T16:27:56.691495Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from IPython.display import Video\nimport random\nimport glob\nimport os\n\n# 1. Ambil semua list video .mp4 yang ada di folder input\nsemua_video = glob.glob('/kaggle/input/**/*.mp4', recursive=True)\n\n# 2. Pilih satu video secara acak\nvideo_pilihan = random.choice(semua_video)\n\n# 3. Ambil nama filenya saja untuk dicek di tabel metadata nanti\nnama_file = os.path.basename(video_pilihan)\n\nprint(f\"🎥 Menampilkan Video: {nama_file}\")\nprint(f\"Path lengkap: {video_pilihan}\")\n\n# 4. Putar videonya\nVideo(video_pilihan, embed=True, width=500)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:36:16.235174Z","iopub.execute_input":"2026-03-31T16:36:16.235810Z","iopub.status.idle":"2026-03-31T16:36:18.714423Z","shell.execute_reply.started":"2026-03-31T16:36:16.235776Z","shell.execute_reply":"2026-03-31T16:36:18.713640Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport glob\n\n# Buat folder baru khusus untuk training massal\nos.makedirs('/kaggle/working/train_dataset_massal', exist_ok=True)\n\ndef extraksi_massal(df, jumlah_video=100, frame_per_video=5):\n    count_total = 0\n    # Ambil video secara acak agar bervariasi\n    df_sample = df.sample(n=jumlah_video).reset_index(drop=True)\n    \n    for i, row in df_sample.iterrows():\n        video_name = row['filename']\n        label = row['label']\n        \n        # Cari lokasi video\n        video_paths = glob.glob(f'/kaggle/input/**/{video_name}', recursive=True)\n        if not video_paths: continue\n        \n        cap = cv2.VideoCapture(video_paths[0])\n        f_count = 0\n        while f_count < frame_per_video:\n            ret, frame = cap.read()\n            if not ret: break\n            \n            # Deteksi wajah (OpenCV)\n            gray = cv2.cvtColor(frame, cv2.COLOR_BGR2GRAY)\n            faces = face_cascade.detectMultiScale(gray, 1.1, 4)\n            \n            for (x, y, w, h) in faces:\n                face_crop = frame[y:y+h, x:x+w]\n                face_crop = cv2.resize(face_crop, (224, 224))\n                \n                # Simpan dengan label di nama file\n                save_name = f\"/kaggle/working/train_dataset_massal/{label}_{video_name}_f{f_count}.jpg\"\n                cv2.imwrite(save_name, cv2.cvtColor(face_crop, cv2.COLOR_RGB2BGR))\n                count_total += 1\n                break\n            f_count += 1\n        cap.release()\n        if i % 10 == 0: print(f\"Progres: {i}/{jumlah_video} video selesai.\")\n\n    print(f\"✅ Selesai! Total {count_total} gambar wajah siap dilatih.\")\n\n# Jalankan ekstraksi untuk 100 video (Bisa kamu tambah jika ingin lebih akurat)\nextraksi_massal(df_meta, jumlah_video=100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T16:43:58.023362Z","iopub.execute_input":"2026-03-31T16:43:58.024214Z","iopub.status.idle":"2026-03-31T16:46:58.784869Z","shell.execute_reply.started":"2026-03-31T16:43:58.024180Z","shell.execute_reply":"2026-03-31T16:46:58.784102Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch.optim as optim\n\n# 1. Load data dari folder massal tadi\ndataset_massal = DeepfakeDataset('/kaggle/working/train_dataset_massal', transform=transform)\ntrain_loader = DataLoader(dataset_massal, batch_size=8, shuffle=True)\n\n# 2. Fungsi Training\ndef train_model(model, loader, epochs=10):\n    criterion = torch.nn.CrossEntropyLoss()\n    optimizer = optim.Adam(model.parameters(), lr=0.0001) # Learning rate lebih kecil agar teliti\n    \n    model.train()\n    for epoch in range(epochs):\n        running_loss = 0.0\n        for images, labels in loader:\n            images, labels = images.to(device), labels.to(device)\n            \n            optimizer.zero_grad()\n            outputs = model(images)\n            loss = criterion(outputs, labels)\n            loss.backward()\n            optimizer.step()\n            \n            running_loss += loss.item()\n        \n        print(f\"Epoch [{epoch+1}/{epochs}] - Rata-rata Loss: {running_loss/len(loader):.4f}\")\n\n# 3. Mulai Training!\nprint(\"🔥 Memulai proses pelatihan massal...\")\ntrain_model(model, train_loader, epochs=10)\nprint(\"✅ AI kamu sekarang sudah sangat terlatih!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T17:29:35.854591Z","iopub.execute_input":"2026-03-31T17:29:35.854868Z","iopub.status.idle":"2026-03-31T17:29:40.686486Z","shell.execute_reply.started":"2026-03-31T17:29:35.854827Z","shell.execute_reply":"2026-03-31T17:29:40.685343Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport random\nimport glob\nimport os\nimport cv2\nfrom PIL import Image\nfrom torchvision import transforms\n\n# --- 1. SETUP PERANGKAT & TRANSFORM ---\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize([0.485, 0.456, 0.406], [0.229, 0.224, 0.225])\n])\n\n# --- 2. DEFINISI MODEL (Agar tidak NameError) ---\nclass CNN_LSTM(nn.Module):\n    def __init__(self, num_classes=2):\n        super(CNN_LSTM, self).__init__()\n        import torchvision.models as models\n        # Gunakan ResNet18 sebagai encoder (CNN)\n        resnet = models.resnet18(pretrained=True)\n        self.cnn = nn.Sequential(*list(resnet.children())[:-1])\n        self.lstm = nn.LSTM(512, 128, batch_first=True)\n        self.fc = nn.Linear(128, num_classes)\n\n    def forward(self, x):\n        batch_size, seq_len, c, h, w = x.size()\n        # Gabungkan batch dan sequence untuk CNN\n        x = x.view(batch_size * seq_len, c, h, w)\n        x = self.cnn(x)\n        x = x.view(batch_size, seq_len, -1)\n        # Masukkan ke LSTM\n        x, _ = self.lstm(x)\n        x = self.fc(x[:, -1, :])\n        return x\n\n# Inisialisasi Model\nmodel = CNN_LSTM().to(device)\n\n# --- 3. FUNGSI DETEKSI ---\ndef deteksi_konten_ai(video_path):\n    model.eval()\n    cap = cv2.VideoCapture(video_path)\n    frames = []\n    total_frames = int(cap.get(cv2.CAP_PROP_FRAME_COUNT))\n    \n    if total_frames <= 5: return print(\"❌ Video terlalu pendek.\")\n\n    for i in range(5):\n        cap.set(cv2.CAP_PROP_POS_FRAMES, (total_frames // 5) * i)\n        ret, frame = cap.read()\n        if not ret: break\n        img = Image.fromarray(cv2.cvtColor(frame, cv2.COLOR_BGR2RGB))\n        frames.append(transform(img))\n    cap.release()\n\n    if len(frames) == 5:\n        input_tensor = torch.stack(frames).unsqueeze(0).to(device)\n        with torch.no_grad():\n            output = model(input_tensor)\n            prob = torch.softmax(output, dim=1)\n            prediction = torch.argmax(prob, dim=1).item()\n            confidence = prob[0][prediction].item() * 100\n            \n        hasil = \"BUATAN AI\" if prediction == 1 else \"ASLI\"\n        print(f\"\\n📢 HASIL ANALISIS SISTEM:\")\n        print(f\"Status    : {hasil}\")\n        print(f\"Keyakinan : {confidence:.2f}%\")\n        print(f\"---------------------------\")\n\n# --- 4. JALANKAN ---\nsemua_video = glob.glob('/kaggle/input/**/*.mp4', recursive=True)\nif len(semua_video) > 0:\n    deteksi_konten_ai(random.choice(semua_video))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-31T18:01:23.373160Z","iopub.execute_input":"2026-03-31T18:01:23.373500Z","iopub.status.idle":"2026-03-31T18:01:33.404532Z","shell.execute_reply.started":"2026-03-31T18:01:23.373474Z","shell.execute_reply":"2026-03-31T18:01:33.403762Z"}},"outputs":[],"execution_count":null}]}