{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19991,"databundleVersionId":1117522,"sourceType":"competition"},{"sourceId":8362912,"sourceType":"datasetVersion","datasetId":4970464}],"dockerImageVersionId":30700,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q efficientnet_pytorch","metadata":{"execution":{"iopub.status.busy":"2024-05-09T07:34:44.810147Z","iopub.execute_input":"2024-05-09T07:34:44.810388Z","iopub.status.idle":"2024-05-09T07:34:49.830497Z","shell.execute_reply.started":"2024-05-09T07:34:44.81036Z","shell.execute_reply":"2024-05-09T07:34:49.829447Z"},"trusted":true},"execution_count":1,"outputs":[{"name":"stdout","text":"\u001b[33mWARNING: Running pip as the 'root' user can result in broken permissions and conflicting behaviour with the system package manager. It is recommended to use a virtual environment instead: https://pip.pypa.io/warnings/venv\u001b[0m\u001b[33m\n\u001b[0m\n\u001b[1m[\u001b[0m\u001b[34;49mnotice\u001b[0m\u001b[1;39;49m]\u001b[0m\u001b[39;49m A new release of pip is available: \u001b[0m\u001b[31;49m23.0.1\u001b[0m\u001b[39;49m -> \u001b[0m\u001b[32;49m24.0\u001b[0m\n\u001b[1m[\u001b[0m\u001b[34;49mnotice\u001b[0m\u001b[1;39;49m]\u001b[0m\u001b[39;49m To update, run: \u001b[0m\u001b[32;49mpip install --upgrade pip\u001b[0m\n","output_type":"stream"}]},{"cell_type":"code","source":"import numpy as np\nfrom PIL import Image\nfrom scipy.spatial import distance\n\ndef validate_vector(u, dtype=None):\n    u = np.asarray(u, dtype=dtype, order='c')\n    if u.ndim == 1:\n        return u\n    raise ValueError(\"Input vector should be 1-D.\")\n    \n\ndef normalize_to_255(arr):\n    # Convert array to numpy array for numerical operations\n    arr = np.array(arr)\n    \n    # Find min and max values in the array\n    min_val = np.min(arr)\n    max_val = np.max(arr)\n    \n    # Normalize each value in the array to range 0-255\n    normalized_arr = ((arr - min_val) / (max_val - min_val)) * 255\n    \n    # Clip values to ensure they are within the range [0, 255]\n    normalized_arr = np.clip(normalized_arr, 0, 255)\n    \n    # Round and convert to integer\n    normalized_arr = normalized_arr.astype(np.uint8)\n    \n    return normalized_arr\n    \ndef change(u):\n    u = validate_vector(u)\n    um = np.mean(u)\n    u = u**4\n    \n    return u\n\ndef check_difference2(cover_image_path, encode_image_path):\n    im1 = Image.open(cover_image_path)\n    im2 = Image.open(encode_image_path)\n    im1_array = np.array(im1).reshape(-1)\n    im2_array = np.array(im2).reshape(-1)\n    \n    im1_array = change(im1_array)\n    im2_array = change(im2_array)\n    \n    # Compute cosine similarity\n    sim = 1 - distance.cosine(im1_array, im2_array)\n    \n    print(\"Cosine similarity: {}\".format(sim))\n\n# Provide paths to your image files\nimage_path_1 = '/kaggle/input/alaska2-image-steganalysis/JUNIWARD/00007.jpg'\nimage_path_2 = '/kaggle/input/alaska2-image-steganalysis/Cover/00007.jpg'\n\ncheck_difference2(image_path_1, image_path_2)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-10T19:31:14.241589Z","iopub.execute_input":"2024-05-10T19:31:14.242017Z","iopub.status.idle":"2024-05-10T19:31:14.277245Z","shell.execute_reply.started":"2024-05-10T19:31:14.241986Z","shell.execute_reply":"2024-05-10T19:31:14.276428Z"},"trusted":true},"execution_count":100,"outputs":[{"name":"stdout","text":"Cosine similarity: 0.9508499398987468\n","output_type":"stream"}]},{"cell_type":"code","source":"import numpy as np\nfrom PIL import Image\nfrom scipy.spatial import distance\n\ndef preprocess_image(image_path):\n    # Open image and convert to grayscale\n    image = Image.open(image_path).convert('L')\n    # Resize the image to a consistent size for processing (e.g., 128x128)\n    image = image.resize((10, 10))\n    # Convert image to numpy array\n    image_array = np.array(image)\n    # Normalize pixel values to range [0, 1]\n    image_array = image_array / 255.0\n    # Flatten the image array to a 1D array\n    image_array_flat = image_array.flatten()\n    return image_array_flat\n\ndef check_difference(cover_image_path, encode_image_path):\n    # Preprocess the cover image\n    cover_image_array = preprocess_image(cover_image_path)\n    # Preprocess the encoded image\n    encode_image_array = preprocess_image(encode_image_path)\n    \n    # Calculate cosine similarity between the two images\n    similarity = 1 - distance.cosine(cover_image_array, encode_image_array)\n    \n    return similarity\n\n# Example usage:\ncover_image_path = '/kaggle/input/alaska2-image-steganalysis/Cover/00001.jpg'\nencode_image_path = '/kaggle/input/alaska2-image-steganalysis/JMiPOD/00001.jpg'\n\nsimilarity_score = check_difference(cover_image_path, encode_image_path)\nprint(\"Cosine similarity:\", similarity_score)\n","metadata":{"execution":{"iopub.status.busy":"2024-05-10T19:35:29.395271Z","iopub.execute_input":"2024-05-10T19:35:29.395677Z","iopub.status.idle":"2024-05-10T19:35:29.419787Z","shell.execute_reply.started":"2024-05-10T19:35:29.395647Z","shell.execute_reply":"2024-05-10T19:35:29.418394Z"},"trusted":true},"execution_count":105,"outputs":[{"name":"stdout","text":"Cosine similarity: 0.9999980264922653\n","output_type":"stream"}]},{"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix\nfrom efficientnet_pytorch import EfficientNet\nfrom albumentations.pytorch import ToTensorV2\nfrom albumentations import (\n    Compose, HorizontalFlip, CLAHE, HueSaturationValue, RandomGamma, OneOf, Resize,\n    ToFloat, ShiftScaleRotate, GridDistortion, RandomRotate90,\n    RGBShift, Blur, MotionBlur, MedianBlur, GaussNoise, CoarseDropout,\n     GaussNoise, OpticalDistortion, RandomSizedCrop, VerticalFlip\n)\nimport os\nimport torch\nimport pandas as pd\nimport numpy as np\nimport random\nimport torch.nn as nn\nimport matplotlib.pyplot as plt\nfrom glob import glob\nimport torchvision\nfrom torch.utils.data import Dataset\nimport time\nfrom tqdm.notebook import tqdm as tqdm_notebook\nfrom tqdm import tqdm as tqdm_short\n# from tqdm import tqdm\nfrom sklearn import metrics\nimport cv2\nimport gc\nimport torch.nn.functional as F\nimport seaborn as sns\nfrom sklearn import metrics","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Alaska2TestDataset(Dataset):\n\n    def __init__(self, df, augmentations=None):\n\n        self.data = df\n        self.augment = augmentations\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        fn = self.data.loc[idx][0]\n        im = cv2.imread(fn)[:, :, ::-1]\n\n        if self.augment:\n            # Apply transformations\n            im = self.augment(image=im)\n\n        return im\n\n\ntest_filenames = sorted(glob(f\"./kaggle/input/alaska2-image-steganalysis/Test/*.jpg\"))\ntest_df = pd.DataFrame({'ImageFileName': list(\n    test_filenames)}, columns=['ImageFileName'])\n\nbatch_size = 16\nnum_workers = 4\n\nAUGMENTATIONS_TEST = Compose([\n    ToFloat(max_value=255),\n    ToTensorV2()\n], p=1)\n\ntest_dataset = Alaska2TestDataset(test_df, augmentations=AUGMENTATIONS_TEST)\ntest_loader = torch.utils.data.DataLoader(test_dataset,\n                                          batch_size=batch_size,\n                                          num_workers=num_workers,\n                                          shuffle=False,\n                                          drop_last=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Net(nn.Module):\n    def __init__(self, num_classes):\n        super().__init__()\n        self.model = EfficientNet.from_pretrained('efficientnet-b0')\n        # 1280 is the number of neurons in last layer. is diff for diff. architecture\n        self.dense_output = nn.Linear(1280, num_classes)\n\n    def forward(self, x):\n        feat = self.model.extract_features(x)\n        feat = F.avg_pool2d(feat, feat.size()[2:]).reshape(-1, 1280)\n        return self.dense_output(feat)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def alaska_weighted_auc(y_true, y_valid):\n    if np.sum(y_true) == 0:\n        # Handle case where there are no positive samples\n        return 0.0\n    \n    tpr_thresholds = [0.0, 0.4, 1.0]\n    weights = [2, 1]\n    \n    fpr, tpr, thresholds = metrics.roc_curve(y_true, y_valid, pos_label=1)\n    \n    areas = np.array(tpr_thresholds[1:]) - np.array(tpr_thresholds[:-1])\n    normalization = np.dot(areas, weights)\n    \n    competition_metric = 0\n    for idx, weight in enumerate(weights):\n        y_min = tpr_thresholds[idx]\n        y_max = tpr_thresholds[idx + 1]\n        mask = (y_min < tpr) & (tpr < y_max)\n        if mask.sum() == 0:\n            continue\n\n        x_padding = np.linspace(fpr[mask][-1], 1, 100)\n        x = np.concatenate([fpr[mask], x_padding])\n        y = np.concatenate([tpr[mask], [y_max] * len(x_padding)])\n        y = y - y_min  # normalize such that curve starts at y=0\n        score = metrics.auc(x, y)\n        submetric = score * weight\n        competition_metric += submetric\n        \n    return competition_metric / normalization","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_names = ['cover','jmipod','juniward','uerd']\nclass_labels = { name: i for i, name in enumerate(class_names)}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = 'cuda'\nmodel = Net(num_classes=len(class_labels)).to(device)\nmodel.load_state_dict(torch.load('/kaggle/input/alaska2-image-steganalysis/Test'))\noptimizer = torch.optim.AdamW(model.parameters(), lr=1e-6)\ncriterion = torch.nn.CrossEntropyLoss()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.eval()\n\npreds = []\ntk0 = tqdm(test_loader)\nwith torch.no_grad():\n    for i, im in enumerate(tk0):\n        inputs = im[\"image\"].to(device, dtype=torch.float)\n        outputs = model(inputs)      \n        preds.extend(F.softmax(outputs, 1).cpu().numpy())\n\npreds = np.array(preds)\nlabels = preds.argmax(1)\nnew_preds = np.zeros((len(preds),))\nnew_preds[labels != 0] = preds[labels != 0, 1:].sum(1)\nnew_preds[labels == 0] = 1 - preds[labels == 0, 0]\n\ntest_df['Id'] = test_df['ImageFileName'].apply(lambda x: x.split(os.sep)[-1])\ntest_df['Label'] = new_preds\n\ntest_df = test_df.drop('ImageFileName', axis=1)\ntest_df.to_csv('submission_eb0.csv', index=False)\nprint(test_df.head())","metadata":{},"execution_count":null,"outputs":[]}]}