{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-08T06:57:12.730575Z","iopub.execute_input":"2022-04-08T06:57:12.73145Z","iopub.status.idle":"2022-04-08T07:03:15.954941Z","shell.execute_reply.started":"2022-04-08T06:57:12.731342Z","shell.execute_reply":"2022-04-08T07:03:15.943392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q efficientnet_pytorch\nfrom efficientnet_pytorch import EfficientNet\nfrom albumentations.pytorch import ToTensorV2\nfrom albumentations import (\n    Compose, HorizontalFlip, CLAHE, HueSaturationValue,\n    RandomBrightness, RandomContrast, RandomGamma, OneOf, Resize,\n    ToFloat, ShiftScaleRotate, GridDistortion, RandomRotate90, Cutout,\n    RGBShift, RandomBrightness, RandomContrast, Blur, MotionBlur, MedianBlur, GaussNoise, CoarseDropout,\n    IAAAdditiveGaussianNoise, GaussNoise, OpticalDistortion, RandomSizedCrop, VerticalFlip\n)\nimport os\nimport torch\nimport pandas as pd\nimport numpy as np\nimport random\nimport torch.nn as nn\nimport matplotlib.pyplot as plt\nfrom glob import glob\nimport torchvision\nfrom torch.utils.data import Dataset\nimport time\nfrom tqdm.notebook import tqdm\n# from tqdm import tqdm\nfrom sklearn import metrics\nimport cv2\nimport gc\nimport torch.nn.functional as F","metadata":{"execution":{"iopub.status.busy":"2022-04-08T07:11:14.202962Z","iopub.execute_input":"2022-04-08T07:11:14.203247Z","iopub.status.idle":"2022-04-08T07:11:21.251355Z","shell.execute_reply.started":"2022-04-08T07:11:14.2032Z","shell.execute_reply":"2022-04-08T07:11:21.250517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nseed = 42\nprint(f'setting everything to seed {seed}')\nrandom.seed(seed)\nos.environ['PYTHONHASHSEED'] = str(seed)\nnp.random.seed(seed)\ntorch.manual_seed(seed)\ntorch.cuda.manual_seed(seed)\ntorch.backends.cudnn.deterministic = True","metadata":{"execution":{"iopub.status.busy":"2022-04-08T07:11:24.394021Z","iopub.execute_input":"2022-04-08T07:11:24.394319Z","iopub.status.idle":"2022-04-08T07:11:24.403611Z","shell.execute_reply.started":"2022-04-08T07:11:24.394284Z","shell.execute_reply":"2022-04-08T07:11:24.402792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ydata_dir = '../input/alaska2-image-steganalysis'\n\nfolder_names = ['JMiPOD/', 'JUNIWARD/', 'UERD/']\nclass_names = ['Normal', 'JMiPOD_75', 'JMiPOD_90', 'JMiPOD_95', \n               'JUNIWARD_75', 'JUNIWARD_90', 'JUNIWARD_95',\n                'UERD_75', 'UERD_90', 'UERD_95']\nclass_labels = { name: i for i, name in enumerate(class_names)}\n","metadata":{"execution":{"iopub.status.busy":"2022-04-08T07:12:07.524655Z","iopub.execute_input":"2022-04-08T07:12:07.524908Z","iopub.status.idle":"2022-04-08T07:12:07.52961Z","shell.execute_reply.started":"2022-04-08T07:12:07.524879Z","shell.execute_reply":"2022-04-08T07:12:07.528927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\ntrain_df = pd.read_csv('../input/alaska2trainvalsplit/alaska2_train_df.csv')\nval_df = pd.read_csv('../input/alaska2trainvalsplit/alaska2_val_df.csv')\n\nprint(train_df.sample(10))\ntrain_df.Label.hist()\nplt.title('Distribution of Classes')","metadata":{"execution":{"iopub.status.busy":"2022-04-08T07:13:05.979502Z","iopub.execute_input":"2022-04-08T07:13:05.980461Z","iopub.status.idle":"2022-04-08T07:13:06.777834Z","shell.execute_reply.started":"2022-04-08T07:13:05.980409Z","shell.execute_reply":"2022-04-08T07:13:06.777174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Alaska2Dataset(Dataset):\n    def __init__(self, df, augmentations=None):\n\n        self.data = df\n        self.augment = augmentations\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        fn, label = self.data.loc[idx]\n        im = cv2.imread(fn)[:, :, ::-1]\n        if self.augment:\n            # Apply transformations\n            im = self.augment(image=im)\n        return im, label\n\n\nimg_size = 512\nAUGMENTATIONS_TRAIN = Compose([\n    VerticalFlip(p=0.5),\n    HorizontalFlip(p=0.5),\n    RandomRotate90(p=0.5),\n    OneOf([\n        Cutout(max_h_size=50,max_w_size=50),\n        CoarseDropout(max_height=50, max_width=50)], p=0.5),\n    ToFloat(max_value=255),\n    ToTensorV2()\n], p=1)\n\n\nAUGMENTATIONS_TEST = Compose([\n    ToFloat(max_value=255),\n    ToTensorV2()\n], p=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-04-08T07:13:25.846588Z","iopub.execute_input":"2022-04-08T07:13:25.846902Z","iopub.status.idle":"2022-04-08T07:13:25.860138Z","shell.execute_reply.started":"2022-04-08T07:13:25.84687Z","shell.execute_reply":"2022-04-08T07:13:25.859297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_df = train_df.sample(64).reset_index(drop=True)\ntrain_dataset = Alaska2Dataset(temp_df, augmentations=AUGMENTATIONS_TEST)\nprint(train_dataset)\nbatch_size = 64\nnum_workers = 0\n\ntemp_loader = torch.utils.data.DataLoader(train_dataset,\n                                          batch_size=batch_size,\n                                          num_workers=num_workers, shuffle=False)\n\n\nimages, labels = next(iter(temp_loader))\nimages = images['image'].permute(0, 2, 3, 1)\nmax_images = 64\ngrid_width = 16\ngrid_height = int(max_images / grid_width)\nfig, axs = plt.subplots(grid_height, grid_width,\n                        figsize=(grid_width+1, grid_height+2))\n\nfor i, (im, label) in enumerate(zip(images, labels)):\n    ax = axs[int(i / grid_width), i % grid_width]\n    ax.imshow(im.squeeze())\n    ax.set_title(str(label.item()))\n    ax.axis('off')\n\nplt.suptitle(\"0: Cover, 1: JMiPOD_75, 2: JMiPOD_90, 3: JMiPOD_95, 4: JUNIWARD_75, 5:JUNIWARD_90,\\n 6: JUNIWARD_95, 7:UERD_75, 8:UERD_90, 9:UERD_95\")\nplt.show()\ndel images\ngc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2022-04-08T07:14:03.405324Z","iopub.execute_input":"2022-04-08T07:14:03.405978Z","iopub.status.idle":"2022-04-08T07:14:09.992278Z","shell.execute_reply.started":"2022-04-08T07:14:03.405942Z","shell.execute_reply":"2022-04-08T07:14:09.99127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = Alaska2Dataset(temp_df, augmentations=AUGMENTATIONS_TRAIN)\nbatch_size = 64\nnum_workers = 0\n\ntemp_loader = torch.utils.data.DataLoader(train_dataset,\n                                          batch_size=batch_size,\n                                          num_workers=num_workers, shuffle=False)\n\n\nimages, labels = next(iter(temp_loader))\nimages = images['image'].permute(0, 2, 3, 1)\nmax_images = 64\ngrid_width = 16\ngrid_height = int(max_images / grid_width)\nfig, axs = plt.subplots(grid_height, grid_width,\n                        figsize=(grid_width+1, grid_height+2))\n\nfor i, (im, label) in enumerate(zip(images, labels)):\n    ax = axs[int(i / grid_width), i % grid_width]\n    ax.imshow(im.squeeze())\n    ax.set_title(str(label.item()))\n    ax.axis('off')\n\nplt.suptitle(\"0: Cover, 1: JMiPOD_75, 2: JMiPOD_90, 3: JMiPOD_95, 4: JUNIWARD_75, 5:JUNIWARD_90,\\n 6: JUNIWARD_95, 7:UERD_75, 8:UERD_90, 9:UERD_95\")\nplt.show()\ndel images, temp_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-04-08T07:14:25.950214Z","iopub.execute_input":"2022-04-08T07:14:25.950939Z","iopub.status.idle":"2022-04-08T07:14:31.374711Z","shell.execute_reply.started":"2022-04-08T07:14:25.950902Z","shell.execute_reply":"2022-04-08T07:14:31.374059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Net(nn.Module):\n    def __init__(self, num_classes):\n        super().__init__()\n        self.model = EfficientNet.from_pretrained('efficientnet-b0')\n        # 1280 is the number of neurons in last layer. is diff for diff. architecture\n        self.dense_output = nn.Linear(1280, num_classes)\n\n    def forward(self, x):\n        feat = self.model.extract_features(x)\n        feat = F.avg_pool2d(feat, feat.size()[2:]).reshape(-1, 1280)\n        return self.dense_output(feat)\n","metadata":{"execution":{"iopub.status.busy":"2022-04-08T07:15:22.247197Z","iopub.execute_input":"2022-04-08T07:15:22.247766Z","iopub.status.idle":"2022-04-08T07:15:22.253741Z","shell.execute_reply.started":"2022-04-08T07:15:22.247726Z","shell.execute_reply":"2022-04-08T07:15:22.252901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 8\nnum_workers = 8\n\ntrain_dataset = Alaska2Dataset(train_df, augmentations=AUGMENTATIONS_TRAIN)\nvalid_dataset = Alaska2Dataset(val_df.sample(1000).reset_index(drop=True), augmentations=AUGMENTATIONS_TEST) #for faster validation sample\n\ntrain_loader = torch.utils.data.DataLoader(train_dataset,\n                                           batch_size=batch_size,\n                                           num_workers=num_workers,\n                                           shuffle=True)\n\nvalid_loader = torch.utils.data.DataLoader(valid_dataset,\n                                           batch_size=batch_size*2,\n                                           num_workers=num_workers,\n                                           shuffle=False)\n\ndevice = 'cuda'\nmodel = Net(num_classes=len(class_labels)).to(device)\n# pretrained model in my pc. now i will train on all images for 2 epochs\nmodel.load_state_dict(torch.load('../input/alaska2trainvalsplit/val_loss_6.08_auc_0.875.pth'))\noptimizer = torch.optim.AdamW(model.parameters(), lr=1e-6)\n","metadata":{"execution":{"iopub.status.busy":"2022-04-08T07:16:05.418816Z","iopub.execute_input":"2022-04-08T07:16:05.419105Z","iopub.status.idle":"2022-04-08T07:16:09.562157Z","shell.execute_reply.started":"2022-04-08T07:16:05.419074Z","shell.execute_reply":"2022-04-08T07:16:09.561423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/anokas/weighted-auc-metric-updated\n\ndef alaska_weighted_auc(y_true, y_valid):\n    tpr_thresholds = [0.0, 0.4, 1.0]\n    weights = [2,   1]\n\n    fpr, tpr, thresholds = metrics.roc_curve(y_true, y_valid, pos_label=1)\n\n    # size of subsets\n    areas = np.array(tpr_thresholds[1:]) - np.array(tpr_thresholds[:-1])\n\n    # The total area is normalized by the sum of weights such that the final weighted AUC is between 0 and 1.\n    normalization = np.dot(areas, weights)\n\n    competition_metric = 0\n    for idx, weight in enumerate(weights):\n        y_min = tpr_thresholds[idx]\n        y_max = tpr_thresholds[idx + 1]\n        mask = (y_min < tpr) & (tpr < y_max)\n        # pdb.set_trace()\n\n        x_padding = np.linspace(fpr[mask][-1], 1, 100)\n\n        x = np.concatenate([fpr[mask], x_padding])\n        y = np.concatenate([tpr[mask], [y_max] * len(x_padding)])\n        y = y - y_min  # normalize such that curve starts at y=0\n        score = metrics.auc(x, y)\n        submetric = score * weight\n        best_subscore = (y_max - y_min) * weight\n        competition_metric += submetric\n\n    return competition_metric / normalization","metadata":{"execution":{"iopub.status.busy":"2022-04-08T07:16:27.978719Z","iopub.execute_input":"2022-04-08T07:16:27.978989Z","iopub.status.idle":"2022-04-08T07:16:27.987634Z","shell.execute_reply.started":"2022-04-08T07:16:27.978957Z","shell.execute_reply":"2022-04-08T07:16:27.98687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"criterion = torch.nn.CrossEntropyLoss()\nnum_epochs = 2\ntrain_loss, val_loss = [], []\n\nfor epoch in range(num_epochs):\n    print('Epoch {}/{}'.format(epoch, num_epochs - 1))\n    print('-' * 10)\n    model.train()\n    running_loss = 0\n    tk0 = tqdm(train_loader, total=int(len(train_loader)))\n    for im, labels in tk0:\n        inputs = im[\"image\"].to(device, dtype=torch.float)\n        labels = labels.to(device, dtype=torch.long)\n        optimizer.zero_grad()\n        outputs = model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item()\n        tk0.set_postfix(loss=(loss.item()))\n\n    epoch_loss = running_loss / (len(train_loader)/batch_size)\n    train_loss.append(epoch_loss)\n    print('Training Loss: {:.8f}'.format(epoch_loss))\n\n    tk1 = tqdm(valid_loader, total=int(len(valid_loader)))\n    model.eval()\n    running_loss = 0\n    y, preds = [], []\n    with torch.no_grad():\n        for (im, labels) in tk1:\n            inputs = im[\"image\"].to(device, dtype=torch.float)\n            labels = labels.to(device, dtype=torch.long)\n            outputs = model(inputs)\n            loss = criterion(outputs, labels)\n            y.extend(labels.cpu().numpy().astype(int))\n            preds.extend(F.softmax(outputs, 1).cpu().numpy())\n            running_loss += loss.item()\n            tk1.set_postfix(loss=(loss.item()))\n\n        epoch_loss = running_loss / (len(valid_loader)/batch_size)\n        val_loss.append(epoch_loss)\n        preds = np.array(preds)\n        # convert multiclass labels to binary class\n        y = np.array(y)\n        labels = preds.argmax(1)\n        for class_label in np.unique(y):\n            idx = y == class_label\n            acc = (labels[idx] == y[idx]).astype(np.float).mean()*100\n            print('accuracy for class', class_names[class_label], 'is', acc)\n        \n        acc = (labels == y).mean()*100\n        new_preds = np.zeros((len(preds),))\n        temp = preds[labels != 0, 1:]\n        new_preds[labels != 0] = temp.sum(1)\n        new_preds[labels == 0] = 1 - preds[labels == 0, 0]\n        y = np.array(y)\n        y[y != 0] = 1\n        auc_score = alaska_weighted_auc(y, new_preds)\n        print(\n            f'Val Loss: {epoch_loss:.3}, Weighted AUC:{auc_score:.3}, Acc: {acc:.3}')\n\n    torch.save(model.state_dict(),\n               f\"epoch_{epoch}_val_loss_{epoch_loss:.3}_auc_{auc_score:.3}.pth\")","metadata":{"execution":{"iopub.status.busy":"2022-04-08T07:16:49.6137Z","iopub.execute_input":"2022-04-08T07:16:49.613957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(15,7))\nplt.plot(train_loss, c='r')\nplt.plot(val_loss, c='b')\nplt.legend(['train_loss', 'val_loss'])\nplt.title('Loss Plot')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint = torch.load('../input/alaska2-public-baseline/best-checkpoint-033epoch.bin')\nnet.load_state_dict(checkpoint['model_state_dict']);\nnet.eval();","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint.keys()\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DatasetSubmissionRetriever(Dataset):\n\n    def __init__(self, image_names, transforms=None):\n        super().__init__()\n        self.image_names = image_names\n        self.transforms = transforms\n\n    def __getitem__(self, index: int):\n        image_name = self.image_names[index]\n        image = cv2.imread(f'{DATA_ROOT_PATH}/Test/{image_name}', cv2.IMREAD_COLOR)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB).astype(np.float32)\n        image /= 255.0\n        if self.transforms:\n            sample = {'image': image}\n            sample = self.transforms(**sample)\n            image = sample['image']\n\n        return image_name, image\n\n    def __len__(self) -> int:\n        return self.image_names.shape[0]","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = DatasetSubmissionRetriever(\n    image_names=np.array([path.split('/')[-1] for path in glob('../input/alaska2-image-steganalysis/Test/*.jpg')]),\n    transforms=get_valid_transforms(),\n)\n\n\ndata_loader = DataLoader(\n    dataset,\n    batch_size=8,\n    shuffle=False,\n    num_workers=2,\n    drop_last=False,\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nresult = {'Id': [], 'Label': []}\nfor step, (image_names, images) in enumerate(data_loader):\n    print(step, end='\\r')\n    \n    y_pred = net(images.cuda())\n    y_pred = 1 - nn.functional.softmax(y_pred, dim=1).data.cpu().numpy()[:,0]\n    \n    result['Id'].extend(image_names)\n    result['Label'].extend(y_pred)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame(result)\nsubmission.to_csv('submission.csv', index=False)\nsubmission.head()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission['Label'].hist(bins=100);\n","metadata":{},"execution_count":null,"outputs":[]}]}