{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport json","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open('/kaggle/input/deepfake-detection-challenge/train_sample_videos/metadata.json', 'r') as file:\n    data = json.load(file)\n\nprint(data)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames = list(data.keys())\n\n# Extract the IDs by removing the '.mp4' from each filename\nids = [filename.split(\".\")[0] for filename in filenames]","metadata":{"execution":{"iopub.status.busy":"2023-08-14T15:01:06.172714Z","iopub.execute_input":"2023-08-14T15:01:06.173217Z","iopub.status.idle":"2023-08-14T15:01:06.177987Z","shell.execute_reply.started":"2023-08-14T15:01:06.173172Z","shell.execute_reply":"2023-08-14T15:01:06.176646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import csv\n\n# Assuming the data string you provided is in the dictionary named `data`\n\n# Extract the filenames (keys of the dictionary)\nfilenames = list(data.keys())\n\n# Extract the IDs by removing the '.mp4' from each filename\nids = [filename.split(\".\")[0] for filename in filenames]\n\n# Map the labels to numbers (FAKE==1, REAL==0)\nlabels = [1 if data[filename]['label'] == 'FAKE' else 0 for filename in filenames]\n\n# Zip together the ids and labels\nrows = zip(ids, labels)\n\n# Write to CSV\nwith open('Labels.csv', 'w', newline='') as csvfile:\n    csv_writer = csv.writer(csvfile)\n    csv_writer.writerow(['ID', 'Label'])  # Writing the headers\n    csv_writer.writerows(rows)  # Writing the data rows\n","metadata":{"execution":{"iopub.status.busy":"2023-08-14T15:03:31.470917Z","iopub.execute_input":"2023-08-14T15:03:31.471323Z","iopub.status.idle":"2023-08-14T15:03:31.477578Z","shell.execute_reply.started":"2023-08-14T15:03:31.471282Z","shell.execute_reply":"2023-08-14T15:03:31.477086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we have a folder 'train_sample_videos' with files aagfhgtpmv.mp4, aapnvogymq.mp4, ....\n# and the csv file 'Labels.csv' with the ids in 1st row as: 'aagfhgtpmv', 'aapnvogymq', ....\n# we will make a code, which will read the video data, and selecting 1 frame per video video instance converting then to jpg format \n# and then save finally into a new folder called 'train_images'\n\nimport cv2\nimport csv\nimport os\n\n# Step 1: Read the CSV file to get the IDs\nwith open('/kaggle/input/tr-sample-labels/Labels.csv', 'r') as csvfile:\n    reader = csv.reader(csvfile)\n    headers = next(reader)  # Skip the headers\n    video_ids = [row[0] for row in reader]\n\n# Create the 'train_images' directory if it doesn't exist\nif not os.path.exists('/kaggle/working/train_images'):\n    os.makedirs('/kaggle/working/train_images')\n\n# Step 2: Loop through each video ID and extract the frame\nfor video_id in video_ids:\n    video_path = os.path.join('/kaggle/input/deepfake-detection-challenge/train_sample_videos/', f'{video_id}.mp4')\n    \n    # Using OpenCV to read the video\n    cap = cv2.VideoCapture(video_path)\n\n    # Check if video opened successfully\n    if not cap.isOpened():\n        print(f\"Error: Couldn't open video file {video_path}\")\n        continue\n\n    # Read the first frame\n    ret, frame = cap.read()\n    \n    if ret:\n        # Construct output image path\n        image_path = os.path.join('train_images', f'{video_id}.jpg')\n        \n        # Save the frame as JPG\n        cv2.imwrite(image_path, frame)\n\n    # Release the video capture object\n    cap.release()\n","metadata":{"execution":{"iopub.status.busy":"2023-08-14T15:20:39.507179Z","iopub.execute_input":"2023-08-14T15:20:39.507487Z","iopub.status.idle":"2023-08-14T15:21:12.354781Z","shell.execute_reply.started":"2023-08-14T15:20:39.507427Z","shell.execute_reply":"2023-08-14T15:21:12.353929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}