{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport tensorflow as tf\nimport tensorflow.keras as keras\nimport cv2\nimport sys\nimport glob\nimport gc\nimport matplotlib.pyplot as plt\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"input_path = '/kaggle/input/deepfake-detection-challenge/'\ntrain_dir = glob.glob(input_path+'train_sample_videos/*.mp4')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"meta = pd.read_json(input_path+'train_sample_videos/metadata.json').T\nmeta.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gc.collect()\nlen(train_dir)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x = meta['label'].value_counts().index\ny = meta['label'].value_counts().values\nplt.bar(x,y)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"REAL = meta[meta['label'] == 'REAL'].index.values\nFAKE_LIST = np.random.choice(meta[meta['label'] == 'FAKE'].index.values,len(REAL))\nfile_list = list(REAL) + list(FAKE_LIST) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"IMG_SHAPE=(229,229,3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"googlenet_base = keras.applications.InceptionV3(input_shape=IMG_SHAPE, include_top=False, weights='imagenet')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# cap = cv2.VideoCapture(train_dir[0])\n# count = 0\n# while count<1:\n#     ret, frame = cap.read()\n#     img = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB)\n#     img = keras.preprocessing.image.img_to_array(img)\n#     img = cv2.resize(img, (229,229))\n#     img = np.expand_dims(img, axis=0)\n#     val = googlenet_base(img)\n#     print(val.shape)\n#     count += 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def creat_features(train_dir, test=False):\n    \n    img_data = {}\n    for files in train_dir:\n        key = files.split('/')[5]\n        \n        if test:\n            data_f = file_list + [ x.split('/')[5] for x in train_dir]\n        else:\n            data_f = file_list\n        if key in data_f:\n            cap = cv2.VideoCapture(files)\n            count = 0\n            if test is False:\n                label = meta.loc[key]['label']\n            print(f'processing video {key}')\n            data = []\n            while count<20:\n                ret, frame = cap.read()\n                if ret == False:\n                    break;\n                img = cv2.cvtColor(frame,cv2.COLOR_BGR2RGB)\n                img = keras.preprocessing.image.img_to_array(img)\n                img = cv2.resize(img, (229,229))\n                #print(img.shape)\n                img = np.expand_dims(img, axis=0)\n                img = googlenet_base(img)\n                data.append(img.numpy())\n                gc.collect()\n                count += 1\n                #print(img.shape)\n            if key in img_data:\n                if test is False:\n                    img_data[key].append([data, label])\n                else:\n                    img_data[key].append(data)\n            else:\n                img_data[key]=[]\n                if test is False:\n                    img_data[key].append([data, label])\n                else:\n                    img_data[key].append(data)\n            cap.release()\n    return img_data","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_features = creat_features(train_dir)\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_df(df_m, test=False):\n    matrix = []\n    label =[]\n    for key in df_m:\n        if test:\n            temp = np.asarray(df_m[key][0])\n            matrix.extend(temp.reshape((temp.shape[0],-1) ))\n        else:\n            temp = np.asarray(df_m[key][0][0])\n            matrix.extend(temp.reshape((temp.shape[0],-1) ))\n            label.extend([df_m[key][0][1]]*20)\n    return matrix, label","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X,y = create_df(train_features)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_dir = glob.glob(input_path+'test_videos/*.mp4')\ntest_features = creat_features(test_dir, test=True)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"X_test,_ = create_df(test_features, test=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.ensemble import RandomForestClassifier","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"m = SVC()\nl = LabelEncoder()\nt = l.fit_transform(y)\nrandom = RandomForestClassifier()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"random.fit(X, t)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predict = random.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"predict.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_df = pd.DataFrame()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_df['filename'] = test_features.keys()\ntemp_df['label'] = np.array(400*[0]).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_df.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from collections import Counter\nstart = 0\ndelta = 20\nfor feat in test_features.keys():\n    end = start+delta\n    results = predict[start:end]\n    temp_df.loc[temp_df['filename']==feat,'label'] = Counter(results).most_common(1)[0][0].astype(int)\n    start = end","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"temp_df.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}