{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":9799555,"sourceType":"datasetVersion","datasetId":6005742},{"sourceId":9799572,"sourceType":"datasetVersion","datasetId":6005756}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":false,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install cmake==3.28.4","metadata":{"execution":{"iopub.status.busy":"2024-11-04T16:16:22.101645Z","iopub.execute_input":"2024-11-04T16:16:22.101934Z","iopub.status.idle":"2024-11-04T16:16:28.687479Z","shell.execute_reply.started":"2024-11-04T16:16:22.101888Z","shell.execute_reply":"2024-11-04T16:16:28.686364Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install dlib==19.9.0","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install face_recognition\n!pip install hdbscan","metadata":{"execution":{"iopub.status.busy":"2024-11-04T16:17:04.060154Z","iopub.execute_input":"2024-11-04T16:17:04.060493Z","iopub.status.idle":"2024-11-04T16:17:10.703428Z","shell.execute_reply.started":"2024-11-04T16:17:04.060438Z","shell.execute_reply":"2024-11-04T16:17:10.702269Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import glob\nimport torch\nimport torchvision\nfrom torchvision import transforms\nfrom torch.utils.data import DataLoader\nfrom torch.utils.data.dataset import Dataset\nimport os\nimport numpy as np\nimport cv2\nimport matplotlib.pyplot as plt\nimport face_recognition\nfrom tqdm.notebook import tqdm\n\n#Check if the file is corrupted or not\ndef validate_video(vid_path,train_transforms):\n      transform = train_transforms\n      count = 20\n      video_path = vid_path\n      frames = []\n      a = int(100/count)\n      first_frame = np.random.randint(0,a)\n      temp_video = video_path.split('/')[-1]\n      for i,frame in enumerate(frame_extract(video_path)):\n        frames.append(transform(frame))\n        if(len(frames) == count):\n          break\n      frames = torch.stack(frames)\n      frames = frames[:count]\n      return frames\n\n#extract a from from video\ndef frame_extract(path):\n  vidObj = cv2.VideoCapture(path) \n  success = 1\n  while success:\n      success, image = vidObj.read()\n      if success:\n          yield image\n\nim_size = 112\nmean = [0.485, 0.456, 0.406]\nstd = [0.229, 0.224, 0.225]\n\ntrain_transforms = transforms.Compose([\n                                        transforms.ToPILImage(),\n                                        transforms.Resize((im_size,im_size)),\n                                        transforms.ToTensor(),\n                                        transforms.Normalize(mean,std)])\n\nvideo_files =  glob.glob('/kaggle/input/deepfake/Celeb_fake_face_only-20241104T042108Z-001/Celeb_fake_face_only/*.mp4')\nvideo_files += glob.glob('/kaggle/input/deepfake/Celeb_real_face_only-20241104T042851Z-001/Celeb_real_face_only/*.mp4')\nvideo_files += glob.glob('/kaggle/input/deepfake/DFDC_FAKE_Face_only_data-20241104T042657Z-001/DFDC_FAKE_Face_only_data/*.mp4')\nvideo_files += glob.glob('/kaggle/input/deepfake/DFDC_REAL_Face_only_data-20241104T042713Z-001/DFDC_REAL_Face_only_data/*.mp4')\nvideo_files += glob.glob('/kaggle/input/deepfake/FF_Face_only_data-20241104T042243Z-001/FF_Face_only_data/*.mp4')\n\nprint(\"Total no of videos :\" , len(video_files))\nprint(video_files)\ncount = 0;\nfor i in video_files:\n  try:\n    count+=1\n    validate_video(i,train_transforms)\n  except:\n    print(\"Number of video processed: \" , count ,\" Remaining : \" , (len(video_files) - count))\n    print(\"Corrupted video is : \" , i)\n    continue\nprint((len(video_files) - count))\n","metadata":{"execution":{"iopub.status.busy":"2024-11-04T19:34:04.075971Z","iopub.execute_input":"2024-11-04T19:34:04.076629Z","iopub.status.idle":"2024-11-04T19:34:54.543719Z","shell.execute_reply.started":"2024-11-04T19:34:04.076386Z","shell.execute_reply":"2024-11-04T19:34:54.542934Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Total videos found:\", len(video_files))","metadata":{"execution":{"iopub.status.busy":"2024-11-04T19:34:54.545306Z","iopub.execute_input":"2024-11-04T19:34:54.545663Z","iopub.status.idle":"2024-11-04T19:34:54.550320Z","shell.execute_reply.started":"2024-11-04T19:34:54.545595Z","shell.execute_reply":"2024-11-04T19:34:54.549639Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#to load preprocessod video to memory\nimport json\nimport copy\nimport random\n\nvideo_files =  glob.glob('/kaggle/input/deepfake/Celeb_fake_face_only-20241104T042108Z-001/Celeb_fake_face_only/*.mp4')\nvideo_files += glob.glob('/kaggle/input/deepfake/Celeb_real_face_only-20241104T042851Z-001/Celeb_real_face_only/*.mp4')\nvideo_files += glob.glob('/kaggle/input/deepfake/DFDC_FAKE_Face_only_data-20241104T042657Z-001/DFDC_FAKE_Face_only_data/*.mp4')\nvideo_files += glob.glob('/kaggle/input/deepfake/DFDC_REAL_Face_only_data-20241104T042713Z-001/DFDC_REAL_Face_only_data/*.mp4')\nvideo_files += glob.glob('/kaggle/input/deepfake/FF_Face_only_data-20241104T042243Z-001/FF_Face_only_data/*.mp4')\n\nrandom.shuffle(video_files)\nrandom.shuffle(video_files)\nframe_count = []\nfor video_file in video_files:\n  cap = cv2.VideoCapture(video_file)\n  if(int(cap.get(cv2.CAP_PROP_FRAME_COUNT))<100):\n    video_files.remove(video_file)\n    continue\n  frame_count.append(int(cap.get(cv2.CAP_PROP_FRAME_COUNT)))\n\nprint(\"frames are \" , frame_count)\nprint(\"Total no of video: \" , len(frame_count))\nprint('Average frame per video:',np.mean(frame_count))","metadata":{"execution":{"iopub.status.busy":"2024-11-04T19:34:54.551842Z","iopub.execute_input":"2024-11-04T19:34:54.552106Z","iopub.status.idle":"2024-11-04T19:35:02.869422Z","shell.execute_reply.started":"2024-11-04T19:34:54.552055Z","shell.execute_reply":"2024-11-04T19:35:02.868383Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# load the video name and labels from csv\nimport torch\nimport torchvision\nfrom torchvision import transforms\nfrom torch.utils.data import DataLoader\nfrom torch.utils.data.dataset import Dataset\nimport os\nimport numpy as np\nimport cv2\nimport matplotlib.pyplot as plt\nimport face_recognition\n\nclass video_dataset(Dataset):\n    def __init__(self,video_names,labels,sequence_length = 60,transform = None):\n        self.video_names = video_names\n        self.labels = labels\n        self.transform = transform\n        self.count = sequence_length\n    def __len__(self):\n        return len(self.video_names)\n    def __getitem__(self,idx):\n        video_path = self.video_names[idx]\n        frames = []\n        a = int(100/self.count)\n        first_frame = np.random.randint(0,a)\n        temp_video = video_path.split('/')[-1]\n        #print(temp_video)\n        label = self.labels.iloc[(labels.loc[labels[\"file\"] == temp_video].index.values[0]),1]\n        if(label == 'FAKE'):\n          label = 0\n        if(label == 'REAL'):\n          label = 1\n        for i,frame in enumerate(self.frame_extract(video_path)):\n          frames.append(self.transform(frame))\n          if(len(frames) == self.count):\n            break\n        frames = torch.stack(frames)\n        frames = frames[:self.count]\n        #print(\"length:\" , len(frames), \"label\",label)\n        return frames,label\n    \n    def frame_extract(self,path):\n      vidObj = cv2.VideoCapture(path) \n      success = 1\n      while success:\n          success, image = vidObj.read()\n          if success:\n              yield image\n\n#plot the image\ndef im_plot(tensor):\n    image = tensor.cpu().numpy().transpose(1,2,0)\n    b,g,r = cv2.split(image)\n    image = cv2.merge((r,g,b))\n    image = image*[0.22803, 0.22145, 0.216989] +  [0.43216, 0.394666, 0.37645]\n    image = image*255.0\n    plt.imshow(image.astype(int))\n    plt.show()\n     ","metadata":{"execution":{"iopub.status.busy":"2024-11-04T19:35:02.874041Z","iopub.execute_input":"2024-11-04T19:35:02.874354Z","iopub.status.idle":"2024-11-04T19:35:02.896392Z","shell.execute_reply.started":"2024-11-04T19:35:02.874303Z","shell.execute_reply":"2024-11-04T19:35:02.895520Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from sklearn.cluster import DBSCAN\n# from hdbscan import HDBSCAN\n# from scipy.fftpack import dct\n# import cv2\n# import numpy as np\n# import torch\n\n# class video_dataset(Dataset):\n#     def __init__(self, video_names, labels, sequence_length=60, transform=None, method='cluster'):\n#         self.video_names = video_names\n#         self.labels = labels\n#         self.transform = transform\n#         self.sequence_length = sequence_length\n#         self.method = method  # Choose between 'cluster', 'entropy', 'motion'\n\n#     def __len__(self):\n#         return len(self.video_names)\n\n#     def __getitem__(self, idx):\n#         video_path = self.video_names[idx]\n#         label = self.get_label(video_path)\n#         frames = list(self.frame_extract(video_path))\n\n#         if self.method == 'cluster':\n#             key_frames = self.select_key_frames_cluster(frames)\n#         else:\n#             raise ValueError(\"Method not supported in this configuration!\")\n\n#         if self.transform:\n#             key_frames = [self.transform(frame) for frame in key_frames]\n\n#         frames = torch.stack(key_frames)\n#         return frames, label\n\n#     def get_label(self, video_path):\n#         temp_video = video_path.split('/')[-1]\n#         label = self.labels.iloc[(labels.loc[labels[\"file\"] == temp_video].index.values[0]), 1]\n#         return 0 if label == 'FAKE' else 1\n\n#     def frame_extract(self, path):\n#         vidObj = cv2.VideoCapture(path)\n#         success = True\n#         while success:\n#             success, image = vidObj.read()\n#             if success:\n#                 yield cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n#     def select_key_frames_cluster(self, frames):\n#         \"\"\"\n#         Cluster candidate frames using HDBSCAN and select one frame per cluster.\n#         \"\"\"\n#         # Step 1: Extract DCT features for clustering\n#         grayscale_frames = [cv2.cvtColor(frame, cv2.COLOR_RGB2GRAY) for frame in frames]\n#         dct_features = [dct(frame.flatten(), norm='ortho')[:256] for frame in grayscale_frames]\n#         dct_features = np.array(dct_features)\n\n#         # Step 2: Apply HDBSCAN clustering\n#         clusterer = HDBSCAN(min_cluster_size=2)\n#         labels = clusterer.fit_predict(dct_features)\n\n#         # Step 3: Select key frames (highest brightness + sharpness score in each cluster)\n#         key_frames = []\n#         for cluster_id in np.unique(labels):\n#             if cluster_id == -1:  # Ignore noise points\n#                 continue\n\n#             cluster_frames = [frames[i] for i in range(len(frames)) if labels[i] == cluster_id]\n#             scores = [\n#                 self.calculate_frame_score(frame) for frame in cluster_frames\n#             ]\n#             best_frame_idx = np.argmax(scores)\n#             key_frames.append(cluster_frames[best_frame_idx])\n\n#         # Step 4: Ensure we have exactly `sequence_length` frames (pad/truncate)\n#         while len(key_frames) < self.sequence_length:\n#             key_frames.append(key_frames[-1])  # Repeat last frame\n#         return key_frames[:self.sequence_length]\n\n#     def calculate_frame_score(self, frame):\n#         \"\"\"\n#         Combine brightness and Laplacian (sharpness) scores.\n#         \"\"\"\n#         brightness = np.mean(frame)\n#         sharpness = cv2.Laplacian(frame, cv2.CV_64F).var()\n#         return brightness + sharpness\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def number_of_real_and_fake_videos(data_list):\n    header_list = [\"file\", \"label\"]\n    lab = pd.read_csv('/kaggle/input/metadata/Gobal_metadata.csv', names=header_list)\n    fake_count = 0\n    real_count = 0\n    real_videos = []\n    fake_videos = []\n\n    # Loop through each video in the provided data list\n    for i in data_list:\n        temp_video = i.split('/')[-1]  # Extract the video filename\n        label = lab.iloc[(lab.loc[lab[\"file\"] == temp_video].index.values[0]), 1]  # Get the label\n\n        # Check if the video is real or fake and update counts and lists\n        if label == 'FAKE':\n            fake_count += 1\n            fake_videos.append(temp_video)  # Append to fake videos list\n        elif label == 'REAL':\n            real_count += 1\n            real_videos.append(temp_video)  # Append to real videos list\n    \n    # Return counts and lists of video filenames\n    return real_count, fake_count, real_videos, fake_videos\n","metadata":{"execution":{"iopub.status.busy":"2024-11-04T21:21:26.072107Z","iopub.execute_input":"2024-11-04T21:21:26.072502Z","iopub.status.idle":"2024-11-04T21:21:26.081383Z","shell.execute_reply.started":"2024-11-04T21:21:26.072441Z","shell.execute_reply":"2024-11-04T21:21:26.080597Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# load the labels and video in data loader\nimport random\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\n\n# video_files = video_fil\n\nheader_list = [\"file\",\"label\"]\nlabels = pd.read_csv('/kaggle/input/metadata/Gobal_metadata.csv',names=header_list)\n#print(labels)\ntrain_videos = video_files[:int(0.8*len(video_files))]\nvalid_videos = video_files[int(0.8*len(video_files)):]\nprint(\"train : \" , len(train_videos))\nprint(\"test : \" , len(valid_videos))\n# train_videos,valid_videos = train_test_split(data,test_size = 0.3)\n# print(train_videos)\n\nreal_train_count, fake_train_count, real_train_videos, fake_train_videos = number_of_real_and_fake_videos(train_videos)\n# Print the counts\nprint(\"TRAIN: Real count:\",real_train_count, \"Fake count:\", fake_train_count)\n\n\nreal_count, fake_count, real_videos, fake_videos = number_of_real_and_fake_videos(valid_videos)\n# Print the counts\nprint(\"TEST: Real count:\", real_count, \"Fake count:\", fake_count)\n\n\nim_size = 112\nmean = [0.485, 0.456, 0.406]\nstd = [0.229, 0.224, 0.225]\n\ntrain_transforms = transforms.Compose([\n                                        transforms.ToPILImage(),\n                                        transforms.Resize((im_size,im_size)),\n                                        transforms.ToTensor(),\n                                        transforms.Normalize(mean,std)])\n\ntest_transforms = transforms.Compose([\n                                        transforms.ToPILImage(),\n                                        transforms.Resize((im_size,im_size)),\n                                        transforms.ToTensor(),\n                                        transforms.Normalize(mean,std)])\n\ntrain_data = video_dataset(train_videos,labels,sequence_length = 80,transform = train_transforms)\n#print(train_data)\nval_data = video_dataset(valid_videos,labels,sequence_length = 80,transform = train_transforms)\ntrain_loader = DataLoader(train_data,batch_size = 4,shuffle = True,num_workers = 4)\nvalid_loader = DataLoader(val_data,batch_size = 4,shuffle = True,num_workers = 4)\nimage,label = train_data[0]\nim_plot(image[0,:,:,:])","metadata":{"execution":{"iopub.status.busy":"2024-11-04T21:23:42.436173Z","iopub.execute_input":"2024-11-04T21:23:42.436541Z","iopub.status.idle":"2024-11-04T21:24:07.691372Z","shell.execute_reply.started":"2024-11-04T21:23:42.436491Z","shell.execute_reply":"2024-11-04T21:24:07.688649Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print the lists of real and fake videos in test\nreal_videos_df = pd.DataFrame(real_videos, columns=['file'])\nprint(real_videos_df)\nfake_videos_df = pd.DataFrame(fake_videos, columns=['file'])\nprint(fake_videos_df)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T21:28:14.075290Z","iopub.execute_input":"2024-11-04T21:28:14.075621Z","iopub.status.idle":"2024-11-04T21:28:14.088338Z","shell.execute_reply.started":"2024-11-04T21:28:14.075574Z","shell.execute_reply":"2024-11-04T21:28:14.087455Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torch import nn\nfrom torchvision import models\n\nclass Model(nn.Module):\n    def __init__(self, num_classes, latent_dim=2048, lstm_layers=1, hidden_dim=2048, bidirectional=False):\n        super(Model, self).__init__()\n        model = models.resnext50_32x4d(pretrained=True)  # Residual Network CNN\n        self.model = nn.Sequential(*list(model.children())[:-2])\n        self.lstm = nn.LSTM(latent_dim, hidden_dim, lstm_layers, bidirectional, batch_first=True)\n        self.relu = nn.LeakyReLU()\n        self.dp = nn.Dropout(0.4)\n        self.linear1 = nn.Linear(2048, num_classes)\n        self.avgpool = nn.AdaptiveAvgPool2d(1)\n        \n        # Attention Layer\n        self.attention = nn.Sequential(\n            nn.Linear(hidden_dim, 128),  # Reduce dimensionality for attention computation\n            nn.Tanh(),                  # Non-linear activation\n            nn.Linear(128, 1),          # Scalar attention score for each frame\n            nn.Softmax(dim=1)           # Normalize scores across all frames\n        )\n\n    def forward(self, x):\n        batch_size, seq_length, c, h, w = x.shape\n        x = x.view(batch_size * seq_length, c, h, w)\n        fmap = self.model(x)\n        x = self.avgpool(fmap)\n        x = x.view(batch_size, seq_length, 2048)\n        \n        x_lstm, _ = self.lstm(x)  # Output shape: (batch_size, seq_length, hidden_dim)\n        \n        # Attention Mechanism\n        attention_scores = self.attention(x_lstm)  # Shape: (batch_size, seq_length, 1)\n        attention_weights = attention_scores / attention_scores.sum(dim=1, keepdim=True)\n        attended_features = (x_lstm * attention_weights).sum(dim=1)  # Weighted sum of frame features\n        \n        # Classification\n        output = self.dp(self.linear1(attended_features))\n        return fmap, output","metadata":{"execution":{"iopub.status.busy":"2024-11-04T21:28:58.729985Z","iopub.execute_input":"2024-11-04T21:28:58.730298Z","iopub.status.idle":"2024-11-04T21:28:58.743172Z","shell.execute_reply.started":"2024-11-04T21:28:58.730253Z","shell.execute_reply":"2024-11-04T21:28:58.741957Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model = Model(2).cuda()\na,b = model(torch.from_numpy(np.empty((1,20,3,112,112))).type(torch.cuda.FloatTensor))","metadata":{"execution":{"iopub.status.busy":"2024-11-04T21:28:59.477814Z","iopub.execute_input":"2024-11-04T21:28:59.478111Z","iopub.status.idle":"2024-11-04T21:29:00.285940Z","shell.execute_reply.started":"2024-11-04T21:28:59.478067Z","shell.execute_reply":"2024-11-04T21:29:00.285248Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torch.autograd import Variable\nimport time\nimport os\nimport sys\nimport os\ndef train_epoch(epoch, num_epochs, data_loader, model, criterion, optimizer):\n    model.train()\n    losses = AverageMeter()\n    accuracies = AverageMeter()\n    t = []\n    for i, (inputs, targets) in enumerate(data_loader):\n        if torch.cuda.is_available():\n            targets = targets.type(torch.cuda.LongTensor)\n            inputs = inputs.cuda()\n        _,outputs = model(inputs)\n        loss  = criterion(outputs,targets.type(torch.cuda.LongTensor))\n        acc = calculate_accuracy(outputs, targets.type(torch.cuda.LongTensor))\n        losses.update(loss.item(), inputs.size(0))\n        accuracies.update(acc, inputs.size(0))\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n        sys.stdout.write(\n                \"\\r[Epoch %d/%d] [Batch %d / %d] [Loss: %f, Acc: %.2f%%]\"\n                % (\n                    epoch,\n                    num_epochs,\n                    i,\n                    len(data_loader),\n                    losses.avg,\n                    accuracies.avg))\n    torch.save(model.state_dict(),'/kaggle/working/checkpoint.pt')\n    return losses.avg,accuracies.avg\ndef test(epoch,model, data_loader ,criterion):\n    print('Testing')\n    model.eval()\n    losses = AverageMeter()\n    accuracies = AverageMeter()\n    pred = []\n    true = []\n    count = 0\n    with torch.no_grad():\n        for i, (inputs, targets) in enumerate(data_loader):\n            if torch.cuda.is_available():\n                targets = targets.cuda().type(torch.cuda.FloatTensor)\n                inputs = inputs.cuda()\n            _,outputs = model(inputs)\n            loss = torch.mean(criterion(outputs, targets.type(torch.cuda.LongTensor)))\n            acc = calculate_accuracy(outputs,targets.type(torch.cuda.LongTensor))\n            _,p = torch.max(outputs,1) \n            true += (targets.type(torch.cuda.LongTensor)).detach().cpu().numpy().reshape(len(targets)).tolist()\n            pred += p.detach().cpu().numpy().reshape(len(p)).tolist()\n            losses.update(loss.item(), inputs.size(0))\n            accuracies.update(acc, inputs.size(0))\n            sys.stdout.write(\n                    \"\\r[Batch %d / %d]  [Loss: %f, Acc: %.2f%%]\"\n                    % (\n                        i,\n                        len(data_loader),\n                        losses.avg,\n                        accuracies.avg\n                        )\n                    )\n        print('\\nAccuracy {}'.format(accuracies.avg))\n    return true,pred,losses.avg,accuracies.avg\nclass AverageMeter(object):\n    \"\"\"Computes and stores the average and current value\"\"\"\n    def __init__(self):\n        self.reset()\n    def reset(self):\n        self.val = 0\n        self.avg = 0\n        self.sum = 0\n        self.count = 0\n\n    def update(self, val, n=1):\n        self.val = val\n        self.sum += val * n\n        self.count += n\n        self.avg = self.sum / self.count\ndef calculate_accuracy(outputs, targets):\n    batch_size = targets.size(0)\n\n    _, pred = outputs.topk(1, 1, True)\n    pred = pred.t()\n    correct = pred.eq(targets.view(1, -1))\n    n_correct_elems = correct.float().sum().item()\n    return 100* n_correct_elems / batch_size\n     ","metadata":{"execution":{"iopub.status.busy":"2024-11-04T21:29:00.287926Z","iopub.execute_input":"2024-11-04T21:29:00.288274Z","iopub.status.idle":"2024-11-04T21:29:00.319410Z","shell.execute_reply.started":"2024-11-04T21:29:00.288183Z","shell.execute_reply":"2024-11-04T21:29:00.318499Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sn\n#Output confusion matrix\ndef print_confusion_matrix(y_true, y_pred):\n    cm = confusion_matrix(y_true, y_pred)\n    print('True positive = ', cm[0][0])\n    print('False positive = ', cm[0][1])\n    print('False negative = ', cm[1][0])\n    print('True negative = ', cm[1][1])\n    print('\\n')\n    df_cm = pd.DataFrame(cm, range(2), range(2))\n    sn.set(font_scale=1.4) # for label size\n    sn.heatmap(df_cm, annot=True, annot_kws={\"size\": 16}) # font size\n    plt.ylabel('Actual label', size = 20)\n    plt.xlabel('Predicted label', size = 20)\n    plt.xticks(np.arange(2), ['Fake', 'Real'], size = 16)\n    plt.yticks(np.arange(2), ['Fake', 'Real'], size = 16)\n    plt.ylim([2, 0])\n    plt.show()\n    calculated_acc = (cm[0][0]+cm[1][1])/(cm[0][0]+cm[0][1]+cm[1][0]+ cm[1][1])\n    accuracy = round(calculated_acc*100,2)\n    print(\"Calculated Accuracy\",calculated_acc*100)\n    return accuracy","metadata":{"execution":{"iopub.status.busy":"2024-11-04T21:43:08.503182Z","iopub.execute_input":"2024-11-04T21:43:08.503582Z","iopub.status.idle":"2024-11-04T21:43:08.516949Z","shell.execute_reply.started":"2024-11-04T21:43:08.503521Z","shell.execute_reply":"2024-11-04T21:43:08.516232Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_loss(train_loss_avg,test_loss_avg,num_epochs):\n  loss_train = train_loss_avg\n  loss_val = test_loss_avg\n  print(num_epochs)\n  epochs = range(1,num_epochs+1)\n  plt.plot(epochs, loss_train, 'g', label='Training loss')\n  plt.plot(epochs, loss_val, 'b', label='validation loss')\n  plt.title('Training and Validation loss')\n  plt.xlabel('Epochs')\n  plt.ylabel('Loss')\n  plt.legend()\n  plt.show()\ndef plot_accuracy(train_accuracy,test_accuracy,num_epochs):\n  loss_train = train_accuracy\n  loss_val = test_accuracy\n  epochs = range(1,num_epochs+1)\n  plt.plot(epochs, loss_train, 'g', label='Training accuracy')\n  plt.plot(epochs, loss_val, 'b', label='validation accuracy')\n  plt.title('Training and Validation accuracy')\n  plt.xlabel('Epochs')\n  plt.ylabel('Accuracy')\n  plt.legend()\n  plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-11-04T21:29:04.961681Z","iopub.execute_input":"2024-11-04T21:29:04.961974Z","iopub.status.idle":"2024-11-04T21:29:04.972078Z","shell.execute_reply.started":"2024-11-04T21:29:04.961933Z","shell.execute_reply":"2024-11-04T21:29:04.971361Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class EarlyStopping:\n    def __init__(self, patience=5, delta=0, path='best_model.pt', verbose=False):\n        \"\"\"\n        Args:\n            patience (int): How many epochs to wait after last improvement.\n            delta (float): Minimum change to qualify as an improvement.\n            path (str): Path to save the best model.\n            verbose (bool): If True, prints a message for each improvement.\n        \"\"\"\n        self.patience = patience\n        self.delta = delta\n        self.path = path\n        self.verbose = verbose\n        self.best_loss = None\n        self.counter = 0\n        self.early_stop = False\n\n    def __call__(self, val_loss, model):\n        if self.best_loss is None:\n            self.best_loss = val_loss\n            self.save_checkpoint(val_loss, model)\n        elif val_loss > self.best_loss + self.delta:\n            self.counter += 1\n            if self.verbose:\n                print(f\"EarlyStopping counter: {self.counter} out of {self.patience}\")\n            if self.counter >= self.patience:\n                self.early_stop = True\n        else:\n            self.best_loss = val_loss\n            self.save_checkpoint(val_loss, model)\n            self.counter = 0\n\n    def save_checkpoint(self, val_loss, model):\n        \"\"\"Saves model when validation loss decreases.\"\"\"\n        if self.verbose:\n            print(f\"Validation loss decreased. Saving model to {self.path}\")\n        torch.save(model.state_dict(), self.path)\n","metadata":{"execution":{"iopub.status.busy":"2024-11-04T21:29:06.254153Z","iopub.execute_input":"2024-11-04T21:29:06.254561Z","iopub.status.idle":"2024-11-04T21:29:06.267577Z","shell.execute_reply.started":"2024-11-04T21:29:06.254504Z","shell.execute_reply":"2024-11-04T21:29:06.266504Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\n# learning rate\nlr = 1e-5  # 0.001\n# number of epochs \nnum_epochs = 30\n\noptimizer = torch.optim.Adam(model.parameters(), lr=lr, weight_decay=1e-5)\n\n# Using CrossEntropyLoss without class weights here\n# criterion = nn.CrossEntropyLoss().cuda()\nclass_weights = torch.from_numpy(np.asarray([1,15])).type(torch.FloatTensor).cuda()\ncriterion = nn.CrossEntropyLoss(weight = class_weights).cuda()\n\ntrain_loss_avg = []\ntrain_accuracy = []\ntest_loss_avg = []\ntest_accuracy = []\n\n# Initialize EarlyStopping\nearly_stopping = EarlyStopping(patience=7, verbose=True, path='/kaggle/working/best_model.pt')\n\nfor epoch in range(1, num_epochs + 1):\n    # Training step\n    l, acc = train_epoch(epoch, num_epochs, train_loader, model, criterion, optimizer)\n    train_loss_avg.append(l)\n    train_accuracy.append(acc)\n    \n    # Validation step\n    true, pred, tl, t_acc = test(epoch, model, valid_loader, criterion)\n    test_loss_avg.append(tl)\n    test_accuracy.append(t_acc)\n    \n    \n    # Early stopping check\n    early_stopping(tl, model)\n    \n    # If early stopping is triggered, break out of the loop\n    if early_stopping.early_stop:\n        print(\"Early stopping triggered.\")\n        break\nplot_loss(train_loss_avg, test_loss_avg, len(train_loss_avg))\nplot_accuracy(train_accuracy, test_accuracy, len(train_accuracy))\n# Confusion matrix\nprint(confusion_matrix(true, pred))\naccuracy = print_confusion_matrix(true, pred)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T21:43:21.076114Z","iopub.execute_input":"2024-11-04T21:43:21.076456Z","iopub.status.idle":"2024-11-04T21:43:51.326190Z","shell.execute_reply.started":"2024-11-04T21:43:21.076396Z","shell.execute_reply":"2024-11-04T21:43:51.324681Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_frames = 80\nfilename = f\"/kaggle/working/model_{accuracy}_acc_{num_frames}_frames_final_data.pt\"","metadata":{"execution":{"iopub.status.busy":"2024-11-04T21:39:11.745133Z","iopub.execute_input":"2024-11-04T21:39:11.745500Z","iopub.status.idle":"2024-11-04T21:39:11.749990Z","shell.execute_reply.started":"2024-11-04T21:39:11.745444Z","shell.execute_reply":"2024-11-04T21:39:11.749120Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the model state\ntorch.save(model.state_dict(), filename)","metadata":{"execution":{"iopub.status.busy":"2024-11-04T21:39:42.009787Z","iopub.execute_input":"2024-11-04T21:39:42.010103Z","iopub.status.idle":"2024-11-04T21:39:42.356447Z","shell.execute_reply.started":"2024-11-04T21:39:42.010058Z","shell.execute_reply":"2024-11-04T21:39:42.355713Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\n\n# Define the model file path\nmodel_file_path = f\"/kaggle/working/model_{accuracy}_acc_{num_frames}_frames_final_data.pt\"\n\n# Define the zip file path\nzip_file_path = \"/kaggle/working/model_80_frame3.zip\"\n\n# Create a zip file containing the model file\nshutil.make_archive(zip_file_path.replace('.zip', ''), 'zip', '/kaggle/working', f'model_{accuracy}_acc_{num_frames}_frames_final_data.pt')\n","metadata":{"execution":{"iopub.status.busy":"2024-11-04T21:05:46.840174Z","iopub.execute_input":"2024-11-04T21:05:46.840544Z","iopub.status.idle":"2024-11-04T21:05:59.886165Z","shell.execute_reply.started":"2024-11-04T21:05:46.840493Z","shell.execute_reply":"2024-11-04T21:05:59.885295Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}