{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":16880,"databundleVersionId":858837,"sourceType":"competition"}],"dockerImageVersionId":29845,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Dropout, GlobalAveragePooling2D\nfrom tensorflow.keras.optimizers import Adam\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T19:55:12.754488Z","iopub.execute_input":"2024-11-18T19:55:12.754769Z","iopub.status.idle":"2024-11-18T19:55:18.023267Z","shell.execute_reply.started":"2024-11-18T19:55:12.75473Z","shell.execute_reply":"2024-11-18T19:55:18.022536Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Шляхи до даних\nTRAIN_DIR = '/kaggle/input/deepfake-detection-challenge/test_videos'\nLABELS_FILE = '/kaggle/input/deepfake-detection-challenge/sample_submission.csv'\nFRAMES_TO_EXTRACT = 10\nIMG_SIZE = 224\nBATCH_SIZE = 32\nEPOCHS = 10","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T20:29:22.703339Z","iopub.execute_input":"2024-11-18T20:29:22.703613Z","iopub.status.idle":"2024-11-18T20:29:22.707408Z","shell.execute_reply.started":"2024-11-18T20:29:22.703571Z","shell.execute_reply":"2024-11-18T20:29:22.70672Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Завантаження міток із CSV\ndef load_labels(labels_file, train_dir):\n    \"\"\"\n    Завантажує мітки та формує повні шляхи до файлів.\n    \"\"\"\n    labels_df = pd.read_csv(labels_file)  # Читаємо файл CSV\n    labels = {}\n    for _, row in labels_df.iterrows():\n        video_path = os.path.join(train_dir, row['filename'])  # Формуємо повний шлях до відео\n        if os.path.exists(video_path):  # Перевіряємо, чи існує файл\n            labels[video_path] = int(row['label'])  # Мітка (0 або 1)\n        else:\n            print(f\"Warning: File {video_path} does not exist.\")\n    return labels\n\n# Витяг кадрів із відео\ndef preprocess_video(video_path, img_size=224, frames_to_extract=5):\n    \"\"\"\n    Зчитує відео, витягує кадри та масштабує до img_size.\n    \"\"\"\n    video = cv2.VideoCapture(video_path)\n    if not video.isOpened():  # Перевіряємо, чи вдалося відкрити відео\n        print(f\"Error: Cannot open video {video_path}\")\n        return np.empty((0, img_size, img_size, 3))\n    \n    frames = []\n    total_frames = int(video.get(cv2.CAP_PROP_FRAME_COUNT))\n    frame_indices = np.linspace(0, total_frames - 1, frames_to_extract, dtype=np.int)\n\n    for i in frame_indices:\n        video.set(cv2.CAP_PROP_POS_FRAMES, i)\n        ret, frame = video.read()\n        if not ret:\n            print(f\"Warning: Frame {i} could not be read from {video_path}\")\n            continue\n        frame = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n        frame = cv2.resize(frame, (img_size, img_size))\n        frames.append(frame)\n    video.release()\n\n    if len(frames) == 0:\n        return np.empty((0, img_size, img_size, 3))\n\n    return np.array(frames)\n\n# Підготовка навчальних даних\ndef prepare_train_data(labels, img_size=224, frames_to_extract=5):\n    \"\"\"\n    Генерує 4-вимірний масив X та 1-вимірний масив y.\n    \"\"\"\n    X, y = [], []\n    for video_path, label in tqdm(labels.items()):\n        frames = preprocess_video(video_path, img_size=img_size, frames_to_extract=frames_to_extract)\n        if frames.size == 0:\n            continue  # Пропускаємо відео, якщо немає кадрів\n        X.extend(frames)\n        y.extend([label] * len(frames))  # Копіюємо мітку для кожного кадру\n    return np.array(X) / 255.0, np.array(y)  # Нормалізуємо пікселі до [0, 1]\n\n# Завантаження міток\nlabels = load_labels(LABELS_FILE, TRAIN_DIR)\n\n# Підготовка даних\nX, y = prepare_train_data(labels, img_size=IMG_SIZE, frames_to_extract=FRAMES_TO_EXTRACT)\n\n#print(f\"Форма X: {X.shape}\")  # Має бути (num_samples, IMG_SIZE, IMG_SIZE, 3)\n#print(f\"Форма y: {y.shape}\")  # Має бути (num_samples,)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T20:34:40.731505Z","iopub.execute_input":"2024-11-18T20:34:40.731765Z","iopub.status.idle":"2024-11-18T21:06:05.713176Z","shell.execute_reply.started":"2024-11-18T20:34:40.731725Z","shell.execute_reply":"2024-11-18T21:06:05.712469Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Спліт даних\nfrom sklearn.model_selection import train_test_split\n\n# Розділення на навчальну і тестову вибірку\ndef split_data(X, y, test_size=0.2, random_state=42):\n    \"\"\"\n    Розділяє дані на навчальну та тестову вибірку.\n    \"\"\"\n    X_train, X_test, y_train, y_test = train_test_split(\n        X, y, test_size=test_size, random_state=random_state\n    )\n    return X_train, X_test, y_train, y_test\n\n# Використання функції\nX_train, X_test, y_train, y_test = split_data(X, y)\n\nprint(f\"Форма X_train: {X_train.shape}, Форма y_train: {y_train.shape}\")\nprint(f\"Форма X_test: {X_test.shape}, Форма y_test: {y_test.shape}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T21:16:06.48444Z","iopub.execute_input":"2024-11-18T21:16:06.484694Z","iopub.status.idle":"2024-11-18T21:16:08.618456Z","shell.execute_reply.started":"2024-11-18T21:16:06.484656Z","shell.execute_reply":"2024-11-18T21:16:08.617755Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Створення моделі\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Dropout, GlobalAveragePooling2D\nfrom tensorflow.keras.optimizers import Adam\n\ndef build_model(img_size=224):\n    base_model = ResNet50(weights=\"imagenet\", include_top=False, input_shape=(img_size, img_size, 3))\n    base_model.trainable = False  # Заморожуємо базову модель\n\n    x = GlobalAveragePooling2D()(base_model.output)\n    x = Dropout(0.5)(x)\n    x = Dense(128, activation=\"relu\")(x)\n    x = Dropout(0.5)(x)\n    output = Dense(1, activation=\"sigmoid\")(x)\n\n    model = Model(inputs=base_model.input, outputs=output)\n    model.compile(optimizer=Adam(learning_rate=1e-4), loss=\"binary_crossentropy\", metrics=[\"accuracy\"])\n    return model\n\n# Навчання моделі\nmodel = build_model()\nhistory = model.fit(X, y, epochs=10, batch_size=32, validation_split=0.2)\n\n# Збереження моделі\nmodel.save(\"/kaggle/working/model.h5\")\nprint(\"Модель збережена!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T21:16:19.065276Z","iopub.execute_input":"2024-11-18T21:16:19.065579Z","iopub.status.idle":"2024-11-18T21:18:09.847961Z","shell.execute_reply.started":"2024-11-18T21:16:19.065521Z","shell.execute_reply":"2024-11-18T21:18:09.846796Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Збереження моделі\nmodel.save(\"/kaggle/working/model.h5\")\nprint(\"Модель збережена!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-18T21:23:14.066431Z","iopub.execute_input":"2024-11-18T21:23:14.066772Z","iopub.status.idle":"2024-11-18T21:23:14.502842Z","shell.execute_reply.started":"2024-11-18T21:23:14.066714Z","shell.execute_reply":"2024-11-18T21:23:14.502139Z"}},"outputs":[],"execution_count":null}]}