{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install pretrainedmodels","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:46:17.384864Z","iopub.execute_input":"2021-10-30T15:46:17.385629Z","iopub.status.idle":"2021-10-30T15:46:28.193943Z","shell.execute_reply.started":"2021-10-30T15:46:17.385525Z","shell.execute_reply":"2021-10-30T15:46:28.19309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport cv2 as cv\nfrom matplotlib import pyplot as plt\nimport matplotlib.image as mpimg\nimport random\nimport time\nfrom PIL import Image, ImageDraw\n\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.optim import SGD\nfrom torch.utils.data.dataset import Dataset\nfrom torchvision import transforms as T\nimport torchvision\nfrom torch.utils.data import DataLoader\nimport torch\nfrom torch import optim\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.datasets import make_classification\nfrom sklearn import metrics, model_selection, preprocessing\nimport albumentations\n\nimport pretrainedmodels\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:46:28.198091Z","iopub.execute_input":"2021-10-30T15:46:28.198319Z","iopub.status.idle":"2021-10-30T15:46:33.188752Z","shell.execute_reply.started":"2021-10-30T15:46:28.198292Z","shell.execute_reply":"2021-10-30T15:46:33.188006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS = 10 # set number of epoch to run \nNUM_WORKERS = 4 # could increase the time to calculate the results \nRANDOM_STATE = 11 # to repproduce results\n\n# path to images\ndata_path = \"../input/fake-video-images-dataset/images_from_video_big/\"\n\nif torch.cuda.is_available():\n    device = \"cuda\"\nelse:\n    device = \"cpu\"\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:46:33.190255Z","iopub.execute_input":"2021-10-30T15:46:33.190553Z","iopub.status.idle":"2021-10-30T15:46:33.244119Z","shell.execute_reply.started":"2021-10-30T15:46:33.190479Z","shell.execute_reply":"2021-10-30T15:46:33.243408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ClassificationDataset:\n    \"\"\"\n    A general classification dataset class\n    \"\"\"\n    def __init__(self, image_paths, targets, resize=None, augmentations=None):\n        \"\"\"\n         image_paths: list of path to images\n         targets: numpy array\n         resize: tuple. Will resizes image if not None\n         augmentations: albumentation augmentations of images\n        \"\"\"\n        self.image_paths = image_paths\n        self.targets = targets\n        self.resize = resize\n        self.augmentations = augmentations\n\n    def __len__(self):\n        \"\"\"\n        Return the total number of samples in the dataset\n        \"\"\"\n        return len(self.image_paths)\n\n    def __getitem__(self, item):\n        \"\"\"\n        Given an index will get image from dataset\n        \"\"\"\n        # PIL to open the image\n        image = Image.open(self.image_paths[item])\n        # convert image to RGB\n        image = image.convert(\"RGB\")\n        # get the from data targets\n        targets = self.targets[item]\n        # resize if Not None\n        if self.resize is not None:\n            image = image.resize((self.resize[1], self.resize[0])) #, resample=Image.BILINEAR)\n        # convert to numpy array\n        image = np.array(image)\n        # if albumentation not None\n        if self.augmentations is not None:\n            augmented = self.augmentations(image=image)\n            image = augmented[\"image\"]\n        # pytorch expects CHW instead of HWC\n        image = np.transpose(image, (2, 0, 1)).astype(np.float32)\n        # Return tensor of images and targets \n        return {\"image\": torch.tensor(image, dtype=torch.float), \n                \"targets\": torch.tensor(targets, dtype=torch.long)}","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:39:46.208502Z","iopub.execute_input":"2021-10-30T15:39:46.209326Z","iopub.status.idle":"2021-10-30T15:39:46.22237Z","shell.execute_reply.started":"2021-10-30T15:39:46.209262Z","shell.execute_reply":"2021-10-30T15:39:46.221629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_model_from():\n    # pretrained models from Pytorch with pretrainedmodels libs\n    model = pretrainedmodels.__dict__[\"resnet18\"](pretrained='imagenet')\n    # add final layers \n    model.last_linear = nn.Sequential(\n        nn.BatchNorm1d(512), \n        nn.Dropout(p=0.25), # \n        nn.Linear(in_features=512, out_features=2048),\n        nn.ReLU(),\n        nn.BatchNorm1d(2048, eps=1e-05, momentum=0.1),\n        nn.Dropout(p=0.5),\n        nn.Linear(in_features=2048, out_features=1))\n\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:46:52.368717Z","iopub.execute_input":"2021-10-30T15:46:52.368966Z","iopub.status.idle":"2021-10-30T15:46:52.3762Z","shell.execute_reply.started":"2021-10-30T15:46:52.368938Z","shell.execute_reply":"2021-10-30T15:46:52.375321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images_path = [data_path + x for x in os.listdir(data_path) if x.endswith('.jpg')]\nimages_label = [int(x[-5:-4]) for x in os.listdir(data_path) if x.endswith('.jpg')]\n\nassert len(images_path) == len(images_label)\n\nimages_path_df = pd.DataFrame(images_path, columns=['image_path'])\nimages_path_df['label'] = images_label\nprint(f'number of images {len(images_path_df)}')\n\nimages = images_path_df.image_path.values.tolist()\ntargets = images_path_df.label.values","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:39:47.346324Z","iopub.execute_input":"2021-10-30T15:39:47.346623Z","iopub.status.idle":"2021-10-30T15:39:49.753159Z","shell.execute_reply.started":"2021-10-30T15:39:47.346592Z","shell.execute_reply":"2021-10-30T15:39:49.751874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get the pretrained model\nmodel = load_model_from()\n\n# move model to device https://pytorch.org/docs/stable/notes/cuda.html\nmodel.to(device)\n# mean and std values of RGB channels for imagenet dataset\nmean = (0.485, 0.456, 0.406)\nstd = (0.229, 0.224, 0.225)\n# albumentations is an image augmentation library\naug = albumentations.Compose([albumentations.Normalize(mean, std,\n                                                       max_pixel_value=255.0, always_apply=True)])\n\n# train_test_split date \ntrain_images, valid_images, train_targets, valid_targets = \\\n        train_test_split(images, targets, stratify=targets, random_state=RANDOM_STATE)\n\n# set train dataset with batch_size\ntrain_dataset = ClassificationDataset(image_paths=train_images, \\\n                                      targets=train_targets, resize=(128, 128), augmentations=aug)\ntrain_loader = torch.utils.data.DataLoader(train_dataset, \\\n                                           batch_size=16, shuffle=True, num_workers=NUM_WORKERS)\n\n# set test dataset with batch_size\nvalid_dataset = ClassificationDataset(image_paths=valid_images, \\\n                                      targets=valid_targets, resize=(128, 128), augmentations=aug)\nvalid_loader = torch.utils.data.DataLoader(valid_dataset, \\\n                                           batch_size=32, shuffle=False, num_workers=NUM_WORKERS)\n\n# simple Adam optimizer\noptimizer = torch.optim.Adam(model.parameters(), lr=5e-4)","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:39:49.756686Z","iopub.execute_input":"2021-10-30T15:39:49.756921Z","iopub.status.idle":"2021-10-30T15:39:50.220566Z","shell.execute_reply.started":"2021-10-30T15:39:49.756893Z","shell.execute_reply":"2021-10-30T15:39:50.219548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"any_random_number = int(np.random.randint(0, 60_000, 1))\nany_random_path = train_loader.dataset.image_paths[any_random_number]\n\ndef draw_picture(any_random_path):\n    any_random_label = any_random_path[-5:-4]\n    im = Image.open(any_random_path)\n    width, height = im.size\n    print(f'Image path {any_random_path}, width {width}, height {height}')\n    print()\n    print(f'Label is {any_random_label}')\n    print()\n    img = mpimg.imread(any_random_path)\n    imgplot = plt.imshow(img)\n    plt.show();\n    \ndraw_picture(any_random_path)","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:39:50.222292Z","iopub.execute_input":"2021-10-30T15:39:50.222632Z","iopub.status.idle":"2021-10-30T15:39:50.518101Z","shell.execute_reply.started":"2021-10-30T15:39:50.222593Z","shell.execute_reply":"2021-10-30T15:39:50.517381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(data_loader, model, optimizer, device):\n    \"\"\"\n    training for one epoch with selected model and params\n     data_loader:  pytorch dataloader\n     model: pytorch model\n     optimizer: optimizer \n     device: cuda/cpu\n    \"\"\"\n    # set training mode \n    model.train()\n    # go over every batch of data in data loader\n    for data in data_loader:\n        inputs = data[\"image\"]\n        targets = data[\"targets\"]\n        # move inputs/targets to cuda/cpu device\n        inputs = inputs.to(device, dtype=torch.float)\n        targets = targets.to(device, dtype=torch.float)\n        # zero grad the optimizer\n        optimizer.zero_grad()\n        # do the forward step of model\n        outputs = model(inputs)\n        # calculate loss\n        loss = nn.BCEWithLogitsLoss()(outputs, targets.view(-1, 1))\n        # backward step the loss\n        loss.backward()\n        # step optimizer\n        optimizer.step()\n        \ndef evaluate(data_loader, model, device):\n    \"\"\"\n    Evaluation for one epoch\n    data_loader: this is the pytorch dataloader\n    model: pytorch model\n    device: cuda/cpu\n    \"\"\"\n    # put model in evaluation mode\n    model.eval()\n    # init lists to store targets and outputs\n    final_targets = []\n    final_outputs = []\n    # no_grad context\n    with torch.no_grad():\n        for data in data_loader:\n            inputs = data[\"image\"]\n            targets = data[\"targets\"]\n            inputs = inputs.to(device, dtype=torch.float)\n            targets = targets.to(device, dtype=torch.float)\n            # generate prediction\n            output = model(inputs)\n            # convert targets and outputs to lists\n            targets = targets.detach().cpu().numpy().tolist()\n            output = output.detach().cpu().numpy().tolist()\n            # extend the original list\n            final_targets.extend(targets)\n            final_outputs.extend(output)\n            \n    return final_outputs, final_targets","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:39:54.906255Z","iopub.execute_input":"2021-10-30T15:39:54.907028Z","iopub.status.idle":"2021-10-30T15:39:54.919464Z","shell.execute_reply.started":"2021-10-30T15:39:54.906982Z","shell.execute_reply":"2021-10-30T15:39:54.918357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for epoch in tqdm(range(EPOCHS)):\n    # train \n    train(train_loader, model, optimizer, device=device)\n    # predict \n    predictions, valid_targets = evaluate(valid_loader, model, device=device)\n    # metrics \n    roc_auc = metrics.roc_auc_score(valid_targets, predictions)\n    print(f\"Epoch={epoch}, Valid ROC AUC={roc_auc}\")","metadata":{"execution":{"iopub.status.busy":"2021-10-30T11:49:47.801227Z","iopub.execute_input":"2021-10-30T11:49:47.801913Z","iopub.status.idle":"2021-10-30T12:23:26.860037Z","shell.execute_reply.started":"2021-10-30T11:49:47.801875Z","shell.execute_reply":"2021-10-30T12:23:26.858128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# save model\n#torch.save(model.state_dict(), './model.bin')\n\n#load model\nmodel_predict = load_model_from()\nmodel_predict.load_state_dict(torch.load('../input/model-file/model.bin', map_location=torch.device('cpu')))\nmodel_predict.eval()","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:47:07.612577Z","iopub.execute_input":"2021-10-30T15:47:07.613099Z","iopub.status.idle":"2021-10-30T15:47:11.123891Z","shell.execute_reply.started":"2021-10-30T15:47:07.613058Z","shell.execute_reply":"2021-10-30T15:47:11.12313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_images_path = '../input/test-images-reface/images_from_video_big_TEST/'\n\ntest_images_path = [test_images_path + x for x in os.listdir(test_images_path) if x.endswith('.jpg')]\n\nassert len(test_images_path) == 35588\n\ntest_images_path[0]","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:49:54.336368Z","iopub.execute_input":"2021-10-30T15:49:54.337125Z","iopub.status.idle":"2021-10-30T15:49:54.37611Z","shell.execute_reply.started":"2021-10-30T15:49:54.337074Z","shell.execute_reply":"2021-10-30T15:49:54.375457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Test_ClassificationDataset:\n    def __init__(self, image_paths, resize=None, augmentations=None):\n        self.image_paths = image_paths\n        self.resize = resize\n        self.augmentations = augmentations\n\n    def __len__(self):\n        return len(self.image_paths)\n\n    def __getitem__(self, item):\n        image = Image.open(self.image_paths[item])\n        image = image.convert(\"RGB\")\n        if self.resize is not None:\n            image = image.resize((self.resize[1], self.resize[0]))\n        image = np.array(image)\n        if self.augmentations is not None:\n            augmented = self.augmentations(image=image)\n            image = augmented[\"image\"]\n        image = np.transpose(image, (2, 0, 1)).astype(np.float32)\n        return {\"image\": torch.tensor(image, dtype=torch.float)}","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:57:02.824355Z","iopub.execute_input":"2021-10-30T15:57:02.825082Z","iopub.status.idle":"2021-10-30T15:57:02.834115Z","shell.execute_reply.started":"2021-10-30T15:57:02.825038Z","shell.execute_reply":"2021-10-30T15:57:02.83302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean = (0.485, 0.456, 0.406)\nstd = (0.229, 0.224, 0.225)\n\naug = albumentations.Compose([albumentations.Normalize(mean, std,\n                                                       max_pixel_value=255.0, always_apply=True)])\n\ntest_images_data = Test_ClassificationDataset(image_paths=test_images_path, resize=(128, 128), augmentations=aug)\n\ntest_images_loader = torch.utils.data.DataLoader(test_images_data, batch_size=16, shuffle=False, num_workers=NUM_WORKERS)\n\ntest_images_loader.dataset.image_paths[0]","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:59:05.731172Z","iopub.execute_input":"2021-10-30T15:59:05.731445Z","iopub.status.idle":"2021-10-30T15:59:05.739437Z","shell.execute_reply.started":"2021-10-30T15:59:05.731414Z","shell.execute_reply":"2021-10-30T15:59:05.738754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_predict.to(device)\n\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2021-10-30T15:59:06.209855Z","iopub.execute_input":"2021-10-30T15:59:06.210323Z","iopub.status.idle":"2021-10-30T15:59:06.220006Z","shell.execute_reply.started":"2021-10-30T15:59:06.210282Z","shell.execute_reply":"2021-10-30T15:59:06.21915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def evaluate_test(data_loader, model, device):\n    num = 0\n    index_count = 0\n    model.eval()\n    result_binary = None\n    final_targets = []\n    with torch.no_grad():\n        for index, data in enumerate(data_loader):\n            inputs = data[\"image\"]\n            inputs = inputs.to(device, dtype=torch.float)\n            output = model(inputs)\n            if result_binary is None:\n                result_binary = output.detach().cpu().numpy().tolist() # [0 if i <= 0 else 1 for i in output]\n            else:\n                result_binary.extend(output.detach().cpu().numpy().tolist()) #[0 if i <= 0 else 1 for i in output]\n            num += 1 \n            index_count += 1\n#             if num > 1:\n#                 break\n#     print(index_count * test_images_loader.batch_size)\n    return result_binary\n\ntotal_result = None\nnumber_of_times = 10\n\nfor nu in range(0, number_of_times, 1):\n    output = evaluate_test(test_images_loader, model_predict, device)\n    if total_result is None:\n        total_result = [item for sublist in output for item in sublist]\n    else:\n        total_result = [x + y for x, y in zip(total_result, [item for sublist in output for item in sublist])]\n        \nlen(total_result)\nassert len(total_result) == 35588","metadata":{"execution":{"iopub.status.busy":"2021-10-30T16:18:31.207365Z","iopub.execute_input":"2021-10-30T16:18:31.207951Z","iopub.status.idle":"2021-10-30T16:30:07.298171Z","shell.execute_reply.started":"2021-10-30T16:18:31.207907Z","shell.execute_reply":"2021-10-30T16:30:07.29739Z"},"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_result_mapped = [0 if i <= 0 else 1 for i in total_result]\nassert len(test_images_path) == len(output)","metadata":{"execution":{"iopub.status.busy":"2021-10-30T16:41:40.023568Z","iopub.execute_input":"2021-10-30T16:41:40.024309Z","iopub.status.idle":"2021-10-30T16:41:40.028767Z","shell.execute_reply.started":"2021-10-30T16:41:40.024255Z","shell.execute_reply":"2021-10-30T16:41:40.028038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_images_no_path = [i.split('/')[4] for i in test_images_path]\ntest_images_no_path[0]","metadata":{"execution":{"iopub.status.busy":"2021-10-30T16:41:40.526419Z","iopub.execute_input":"2021-10-30T16:41:40.527293Z","iopub.status.idle":"2021-10-30T16:41:40.535298Z","shell.execute_reply.started":"2021-10-30T16:41:40.527245Z","shell.execute_reply":"2021-10-30T16:41:40.534272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_video_name = [i.split('_')[0] for i in test_images_no_path]\n\nassert len(test_video_name) == 35588\n\ntest_video_name[0]","metadata":{"execution":{"iopub.status.busy":"2021-10-30T16:41:41.039298Z","iopub.execute_input":"2021-10-30T16:41:41.039585Z","iopub.status.idle":"2021-10-30T16:41:41.043246Z","shell.execute_reply.started":"2021-10-30T16:41:41.039552Z","shell.execute_reply":"2021-10-30T16:41:41.042546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_df = pd.DataFrame(test_video_name, columns=['filename'])\n\nresult_df['label'] = total_result_mapped\n\nresult_df.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-10-30T16:41:44.366959Z","iopub.execute_input":"2021-10-30T16:41:44.367219Z","iopub.status.idle":"2021-10-30T16:41:44.398188Z","shell.execute_reply.started":"2021-10-30T16:41:44.367187Z","shell.execute_reply":"2021-10-30T16:41:44.397415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# how results are distiriubted  \nresult_df.groupby('filename')['label'].sum().value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-10-30T16:41:51.291355Z","iopub.execute_input":"2021-10-30T16:41:51.291922Z","iopub.status.idle":"2021-10-30T16:41:51.330368Z","shell.execute_reply.started":"2021-10-30T16:41:51.291881Z","shell.execute_reply":"2021-10-30T16:41:51.329676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_calculated_df = pd.DataFrame(result_df.groupby('filename')['label'].mean()).reset_index()\nresult_calculated_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2021-10-30T16:41:55.379211Z","iopub.execute_input":"2021-10-30T16:41:55.379471Z","iopub.status.idle":"2021-10-30T16:41:55.424121Z","shell.execute_reply.started":"2021-10-30T16:41:55.379438Z","shell.execute_reply":"2021-10-30T16:41:55.422444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result_calculated_updated.to_csv('sample_submission_08.csv', index_label=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-30T16:42:21.727698Z","iopub.execute_input":"2021-10-30T16:42:21.728245Z","iopub.status.idle":"2021-10-30T16:42:21.796101Z","shell.execute_reply.started":"2021-10-30T16:42:21.728206Z","shell.execute_reply":"2021-10-30T16:42:21.795186Z"},"trusted":true},"execution_count":null,"outputs":[]}]}