{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# from transformers import AutoFeatureExtractor, AutoModel\n","metadata":{"execution":{"iopub.status.busy":"2023-07-26T16:34:57.976358Z","iopub.execute_input":"2023-07-26T16:34:57.976774Z","iopub.status.idle":"2023-07-26T16:35:00.147620Z","shell.execute_reply.started":"2023-07-26T16:34:57.976743Z","shell.execute_reply":"2023-07-26T16:35:00.146466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Our model which is trained on badlad ***","metadata":{}},{"cell_type":"code","source":"from transformers import AutoFeatureExtractor, AutoModel\nimport torch\n\n# Path to the downloaded .pth model checkpoint\nmodel_ckpt_path = \"RadAlienware/layoutlmv3\"\n\n# Load the model and feature extractor\nextractor = AutoFeatureExtractor.from_pretrained(model_ckpt_path)\nmodel = AutoModel.from_pretrained(model_ckpt_path)\n","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:06:18.312840Z","iopub.execute_input":"2023-08-02T07:06:18.313263Z","iopub.status.idle":"2023-08-02T07:06:46.856662Z","shell.execute_reply.started":"2023-08-02T07:06:18.313229Z","shell.execute_reply":"2023-08-02T07:06:46.855508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**hugging face model**","metadata":{}},{"cell_type":"code","source":"# from transformers import AutoFeatureExtractor, AutoModel\n# import torch\n\n# # Path to the downloaded .pth model checkpoint\n# model_ckpt_path = \"DunnBC22/dit-base-Document_Classification-RVL_CDIP\"\n\n# # Load the model and feature extractor\n# extractor = AutoFeatureExtractor.from_pretrained(model_ckpt_path)\n# model = AutoModel.from_pretrained(model_ckpt_path)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-27T13:03:27.196077Z","iopub.execute_input":"2023-07-27T13:03:27.197170Z","iopub.status.idle":"2023-07-27T13:03:45.277163Z","shell.execute_reply.started":"2023-07-27T13:03:27.197117Z","shell.execute_reply":"2023-07-27T13:03:45.276367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from transformers import AutoFeatureExtractor, AutoModel\n# import torch\n\n# # Path to the downloaded .pth model checkpoint\n# model_ckpt_path = \"microsoft/dit-large\"\n\n# # Load the model and feature extractor\n# extractor = AutoFeatureExtractor.from_pretrained(model_ckpt_path)\n# model = AutoModel.from_pretrained(model_ckpt_path)","metadata":{"execution":{"iopub.status.busy":"2023-07-26T22:28:45.964753Z","iopub.execute_input":"2023-07-26T22:28:45.965499Z","iopub.status.idle":"2023-07-26T22:28:55.293692Z","shell.execute_reply.started":"2023-07-26T22:28:45.965434Z","shell.execute_reply":"2023-07-26T22:28:55.292526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%capture\nimport sys, os, distutils.core\n# Note: This is a faster way to install detectron2 in Colab, but it does not include all functionalities (e.g. compiled operators).\n# See https://detectron2.readthedocs.io/tutorials/install.html for full installation instructions\n!git clone 'https://github.com/facebookresearch/detectron2'\ndist = distutils.core.run_setup(\"./detectron2/setup.py\")\n!python -m pip install {' '.join([f\"'{x}'\" for x in dist.install_requires])}\nsys.path.insert(0, os.path.abspath('./detectron2'))","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:06:50.801183Z","iopub.execute_input":"2023-08-02T07:06:50.801589Z","iopub.status.idle":"2023-08-02T07:07:05.683761Z","shell.execute_reply.started":"2023-08-02T07:06:50.801554Z","shell.execute_reply":"2023-08-02T07:07:05.682128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pathlib import Path\n\n# TRAIN_IMG_DIR = Path(\"/kaggle/input/dlsprint2/badlad/images/train\")\n\n# TRAIN_COCO_PATH = Path(\"/kaggle/input/dlsprint2/badlad/labels/coco_format/train/badlad-train-coco.json\")\n\nTEST_IMG_DIR = Path(\"/kaggle/input/dlsprint2/badlad/images/test\")\n\nTEST_METADATA_PATH = Path(\"/kaggle/input/dlsprint2/badlad/badlad-test-metadata.json\")\n","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:07:05.686654Z","iopub.execute_input":"2023-08-02T07:07:05.687175Z","iopub.status.idle":"2023-08-02T07:07:05.694019Z","shell.execute_reply.started":"2023-08-02T07:07:05.687124Z","shell.execute_reply":"2023-08-02T07:07:05.692931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport pandas as pd\n\n# Define the path to your image folder\nimage_folder_path = \"/kaggle/input/badlad-inference-pseudolabels-p6/epaper\"\n\n# Create empty lists to store the file names, heights, widths, and image IDs\nfile_names = []\nheights = []\nwidths = []\nimage_ids = []\n\n# Iterate through the images in the folder, read their metadata, and append the data to the respective lists\nfor idx, file_name in enumerate(os.listdir(image_folder_path), start=1):\n    # Get the full path to the image\n    image_path = os.path.join(image_folder_path, file_name)\n\n    # Open the image using PIL\n    image = Image.open(image_path)\n\n    # Get the height and width of the image\n    height, width = image.size\n\n    # Append the data to the lists\n    file_names.append(file_name)\n    heights.append(height)\n    widths.append(width)\n    image_ids.append(idx)\n\n# Create a pandas DataFrame using the lists\ndata = {\n    \"image_id\": image_ids,\n    \"file_name\": file_names,\n    \"height\": heights,\n    \"width\": widths\n}\n\nimage_df = pd.DataFrame(data)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:07:24.996228Z","iopub.execute_input":"2023-08-02T07:07:24.996636Z","iopub.status.idle":"2023-08-02T07:14:39.331035Z","shell.execute_reply.started":"2023-08-02T07:07:24.996601Z","shell.execute_reply":"2023-08-02T07:14:39.328523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(image_df)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:15:19.394755Z","iopub.execute_input":"2023-08-02T07:15:19.396304Z","iopub.status.idle":"2023-08-02T07:15:19.446864Z","shell.execute_reply.started":"2023-08-02T07:15:19.396253Z","shell.execute_reply":"2023-08-02T07:15:19.445101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport pandas as pd\n\n# Define the path to your image folder\nimage_folder_path = \"/kaggle/input/badlad-inference-pseudolabels-p6/epaper\"\n\n# Create empty lists to store the file names, heights, widths, and image IDs\nfile_names = []\nheights = []\nwidths = []\nimage_ids = []\n\n# Iterate through the images in the folder, read their metadata, and append the data to the respective lists\nfor idx, file_name in enumerate(os.listdir(image_folder_path), start=1):\n    # Get the full path to the image\n    image_path = os.path.join(image_folder_path, file_name)\n\n    try:\n        # Open the image using PIL\n        image = Image.open(image_path)\n\n        # Get the height and width of the image\n        height, width = image.size\n\n        # Append the data to the lists\n        file_names.append(file_name)\n        heights.append(height)\n        widths.append(width)\n        image_ids.append(idx)\n    except IOError:\n        # If an error occurs while opening the image, skip this file and continue to the next one\n        print(f\"Error opening image: {image_path}\")\n        continue\n\n# Create a pandas DataFrame using the lists\ndata = {\n    \"image_id\": image_ids,\n    \"file_name\": file_names,\n    \"height\": heights,\n    \"width\": widths\n}\n\nimage_df = pd.DataFrame(data)\n\n# Now you have a DataFrame containing the image metadata, including image ID, file name, height, and width.\n\n\n","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:30:46.178367Z","iopub.execute_input":"2023-08-02T07:30:46.179872Z","iopub.status.idle":"2023-08-02T07:41:34.216246Z","shell.execute_reply.started":"2023-08-02T07:30:46.179821Z","shell.execute_reply":"2023-08-02T07:41:34.215058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(image_df)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:44:07.754496Z","iopub.execute_input":"2023-08-02T07:44:07.755008Z","iopub.status.idle":"2023-08-02T07:44:07.797881Z","shell.execute_reply.started":"2023-08-02T07:44:07.754969Z","shell.execute_reply":"2023-08-02T07:44:07.795325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # detectron2\n# from detectron2.utils.memory import retry_if_cuda_oom\n# from detectron2.utils.logger import setup_logger\n# from detectron2.checkpoint import DetectionCheckpointer\n# from detectron2.modeling import build_model\n# from detectron2.evaluation import COCOEvaluator, inference_on_dataset\n# import detectron2.data.transforms as T\n# from detectron2.data import detection_utils as utils\n# from detectron2.data import DatasetCatalog, MetadataCatalog, build_detection_test_loader, build_detection_train_loader, DatasetMapper\n# from detectron2.utils.visualizer import Visualizer\n# from detectron2.structures import BoxMode\n# from detectron2.engine import DefaultPredictor, DefaultTrainer\n# from detectron2.config import get_cfg\n# from detectron2 import model_zoo\n\nimport pandas as pd\nimport numpy as np\nfrom tqdm.notebook import tqdm  # progress bar\nimport matplotlib.pyplot as plt\nimport json\nimport cv2\nimport copy\nfrom typing import Optional\n\nfrom IPython.display import FileLink\n\n# torch\nimport torch\n\nimport gc\n\nimport warnings\n# Ignore \"future\" warnings and Data-Frame-Slicing warnings.\nwarnings.filterwarnings('ignore')\n\n# setup_logger()","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:44:16.342842Z","iopub.execute_input":"2023-08-02T07:44:16.343327Z","iopub.status.idle":"2023-08-02T07:44:16.625592Z","shell.execute_reply.started":"2023-08-02T07:44:16.343290Z","shell.execute_reply":"2023-08-02T07:44:16.624623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# with TRAIN_COCO_PATH.open() as f:\n#     train_dict = json.load(f)\n\nwith TEST_METADATA_PATH.open() as f:\n    test_dict = json.load(f)\n\nprint(\"#### LABELS AND METADATA LOADED ####\")","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:44:22.833514Z","iopub.execute_input":"2023-08-02T07:44:22.833916Z","iopub.status.idle":"2023-08-02T07:44:22.909762Z","shell.execute_reply.started":"2023-08-02T07:44:22.833884Z","shell.execute_reply":"2023-08-02T07:44:22.908519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def organize_coco_data(data_dict: dict) -> tuple[list[str], list[dict], list[dict]]:\n    thing_classes: list[str] = []\n\n    # Map Category Names to IDs\n    for cat in data_dict['categories']:\n        thing_classes.append(cat['name'])\n\n    # Images\n    images_metadata: list[dict] = data_dict['images']\n\n    # Convert COCO annotations to detectron2 annotations format\n    data_annotations = []\n    for ann in data_dict['annotations']:\n        # coco format -> detectron2 format\n        annot_obj = {\n            # Annotation ID\n            \"id\": ann['id'],\n\n            # Segmentation Polygon (x, y) coords\n            \"gt_masks\": ann['segmentation'],\n\n            # Image ID for this annotation (Which image does this annotation belong to?)\n            \"image_id\": ann['image_id'],\n\n            # Category Label (0: paragraph, 1: text box, 2: image, 3: table)\n            \"category_id\": ann['category_id'],\n\n            \"x_min\": ann['bbox'][0],  # left\n            \"y_min\": ann['bbox'][1],  # top\n            \"x_max\": ann['bbox'][0] + ann['bbox'][2],  # left+width\n            \"y_max\": ann['bbox'][1] + ann['bbox'][3]  # top+height\n        }\n        data_annotations.append(annot_obj)\n\n    return thing_classes, images_metadata, data_annotations","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:44:23.910143Z","iopub.execute_input":"2023-08-02T07:44:23.910785Z","iopub.status.idle":"2023-08-02T07:44:23.919839Z","shell.execute_reply.started":"2023-08-02T07:44:23.910751Z","shell.execute_reply":"2023-08-02T07:44:23.918519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# thing_classes, images_metadata, data_annotations = organize_coco_data(\n#     train_dict\n# )\n\nthing_classes_test, images_metadata_test, _ = organize_coco_data(\n    test_dict\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:44:25.010821Z","iopub.execute_input":"2023-08-02T07:44:25.011919Z","iopub.status.idle":"2023-08-02T07:44:25.016682Z","shell.execute_reply.started":"2023-08-02T07:44:25.011878Z","shell.execute_reply":"2023-08-02T07:44:25.015355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:44:26.278297Z","iopub.execute_input":"2023-08-02T07:44:26.278738Z","iopub.status.idle":"2023-08-02T07:44:26.283758Z","shell.execute_reply.started":"2023-08-02T07:44:26.278704Z","shell.execute_reply":"2023-08-02T07:44:26.282792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_metadata = pd.DataFrame(images_metadata_test)\ntest_metadata = test_metadata[['id', 'file_name', 'width', 'height']]\ntest_metadata = test_metadata.rename(columns={\"id\": \"image_id\"})\nprint(\"test_metadata size=\", len(test_metadata))\ntest_metadata.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:44:27.209823Z","iopub.execute_input":"2023-08-02T07:44:27.210302Z","iopub.status.idle":"2023-08-02T07:44:27.272377Z","shell.execute_reply.started":"2023-08-02T07:44:27.210265Z","shell.execute_reply":"2023-08-02T07:44:27.270991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# from IPython.display import display\n# from PIL import Image\n","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:44:28.150034Z","iopub.execute_input":"2023-08-02T07:44:28.150416Z","iopub.status.idle":"2023-08-02T07:44:28.154934Z","shell.execute_reply.started":"2023-08-02T07:44:28.150387Z","shell.execute_reply":"2023-08-02T07:44:28.153424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def display_image(image_path):\n#     img = Image.open(image_path)\n#     display(img)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:18.290571Z","iopub.execute_input":"2023-07-28T17:08:18.290935Z","iopub.status.idle":"2023-07-28T17:08:18.294269Z","shell.execute_reply.started":"2023-07-28T17:08:18.290907Z","shell.execute_reply":"2023-07-28T17:08:18.293492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n\n# # Assuming you have the DataFrame with image data called 'dataset'\n# # dataset = pd.read_csv('your_file.csv')\n\n# # Define the number of samples you want to select\n# num_samples = 3000\n\n# # Define the random seed for reproducibility\n# seed = 42\n\n# # Shuffle the DataFrame using the random seed\n# shuffled_dataset = test_metadata.sample(frac=1, random_state=seed)\n\n# # Select the first 'num_samples' rows from the shuffled DataFrame\n# candidate_subset = shuffled_dataset.head(num_samples)\n\n# # Now 'candidate_subset' contains the randomly selected 100 images\n# print(candidate_subset)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:18.768454Z","iopub.execute_input":"2023-07-28T17:08:18.769077Z","iopub.status.idle":"2023-07-28T17:08:18.773666Z","shell.execute_reply.started":"2023-07-28T17:08:18.769041Z","shell.execute_reply":"2023-07-28T17:08:18.772754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import torchvision.transforms as T\n\n\n# # Data transformation chain.\n# transformation_chain = T.Compose(\n#     [\n#         # We first resize the input image to 256x256 and then we take center crop.\n#         T.Resize(int((256 / 224) * extractor.size[\"height\"])),\n#         T.CenterCrop(extractor.size[\"height\"]),\n#         T.ToTensor(),\n#         T.Normalize(mean=extractor.image_mean, std=extractor.image_std),\n#     ]\n# )","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:19.127642Z","iopub.execute_input":"2023-07-28T17:08:19.128399Z","iopub.status.idle":"2023-07-28T17:08:19.133591Z","shell.execute_reply.started":"2023-07-28T17:08:19.128352Z","shell.execute_reply":"2023-07-28T17:08:19.132361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Initialize your model and transformation_chain as needed\n\n# # Set the device for the model\n# device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n# model = model.to(device)\n# image_dir = \"/kaggle/input/dlsprint2/badlad/images/test\"\n\n# # Create a new empty column for embeddings in the DataFrame\n# candidate_subset[\"embeddings\"] = None\n\n# # Define the function to compute embeddings for a single image\n# def compute_embeddings(image_filename):\n#     image_path = os.path.join(image_dir, image_filename)\n#     image = Image.open(image_path)\n#     image=image.convert(\"RGB\")\n#     image_transformed = transformation_chain(image)\n#     new_batch = {\"pixel_values\": image_transformed.unsqueeze(0).to(device)}\n#     with torch.no_grad():\n#         embeddings = model(**new_batch).last_hidden_state[:, 0].cpu()\n#         return embeddings.tolist()\n# #     return embeddings[0].tolist()\n\n# # Apply the compute_embeddings function to each image in the DataFrame\n# candidate_subset[\"embeddings\"] = candidate_subset[\"file_name\"].apply(compute_embeddings)\n\n# # Now 'candidate_subset' contains the embeddings for each image\n# print(candidate_subset)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:19.496687Z","iopub.execute_input":"2023-07-28T17:08:19.497129Z","iopub.status.idle":"2023-07-28T17:08:19.502263Z","shell.execute_reply.started":"2023-07-28T17:08:19.497093Z","shell.execute_reply":"2023-07-28T17:08:19.501436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(candidate_subset[\"embeddings\"])","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:20.246972Z","iopub.execute_input":"2023-07-28T17:08:20.247397Z","iopub.status.idle":"2023-07-28T17:08:20.251200Z","shell.execute_reply.started":"2023-07-28T17:08:20.247366Z","shell.execute_reply":"2023-07-28T17:08:20.250352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# two_dimensional_list = candidate_subset[\"embeddings\"].tolist()\n\n# # Now 'two_dimensional_list' is a 2D list containing the embeddings from the DataFrame\n# # print(two_dimensional_list)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:20.616149Z","iopub.execute_input":"2023-07-28T17:08:20.616542Z","iopub.status.idle":"2023-07-28T17:08:20.621017Z","shell.execute_reply.started":"2023-07-28T17:08:20.616509Z","shell.execute_reply":"2023-07-28T17:08:20.619734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n# import pandas as pd\n\n# # Convert the list of lists to a numpy array\n# embeddings_array = np.array(two_dimensional_list)\n\n# # Or, convert the list of lists to a pandas DataFrame\n# embeddings_df1 = pd.DataFrame(two_dimensional_list)\n# print(embeddings_df1)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:21.244984Z","iopub.execute_input":"2023-07-28T17:08:21.245442Z","iopub.status.idle":"2023-07-28T17:08:21.250493Z","shell.execute_reply.started":"2023-07-28T17:08:21.245407Z","shell.execute_reply":"2023-07-28T17:08:21.249243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd\n# import torch\n\n# # Assuming 'df' is your DataFrame with the \"embeddings\" column\n# # The \"embeddings\" column contains lists of lists in each row\n\n# # Convert the \"embeddings\" column to a tensor with the desired shape\n# candidate_subset[\"embeddings\"] = candidate_subset[\"embeddings\"].apply(lambda x: torch.tensor(x, dtype=torch.float64))\n\n# # Now 'df[\"embeddings\"]' contains tensors with the desired shape\n# print(candidate_subset[\"embeddings\"])\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:21.775730Z","iopub.execute_input":"2023-07-28T17:08:21.776167Z","iopub.status.idle":"2023-07-28T17:08:21.781177Z","shell.execute_reply.started":"2023-07-28T17:08:21.776131Z","shell.execute_reply":"2023-07-28T17:08:21.779890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n\n# all_candidate_embeddings = np.array(two_dimensional_list)\n# all_candidate_embeddings = torch.from_numpy(all_candidate_embeddings)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:22.188128Z","iopub.execute_input":"2023-07-28T17:08:22.188507Z","iopub.status.idle":"2023-07-28T17:08:22.192650Z","shell.execute_reply.started":"2023-07-28T17:08:22.188478Z","shell.execute_reply":"2023-07-28T17:08:22.191774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### print(all_candidate_embeddings)","metadata":{"execution":{"iopub.status.busy":"2023-07-27T13:21:41.842938Z","iopub.execute_input":"2023-07-27T13:21:41.843354Z","iopub.status.idle":"2023-07-27T13:21:41.853021Z","shell.execute_reply.started":"2023-07-27T13:21:41.843322Z","shell.execute_reply":"2023-07-27T13:21:41.851120Z"}}},{"cell_type":"code","source":"# import pandas as pd\n# import torch\n\n# # Assuming 'all_candidate_embeddings' is your tensor with shape (100, 1, 768)\n\n# # Reshape the tensor to have shape (100, 768)\n# reshaped_embeddings = all_candidate_embeddings.view(3000, 768)\n\n# # Convert the reshaped tensor to a pandas DataFrame\n# column_names = [f\"embedding_{i}\" for i in range(reshaped_embeddings.shape[1])]\n# all_candidate_embeddings_df = pd.DataFrame(reshaped_embeddings, columns=column_names)\n\n# # Now 'all_candidate_embeddings_df' is a DataFrame with the desired shape\n# print(all_candidate_embeddings_df)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:23.884636Z","iopub.execute_input":"2023-07-28T17:08:23.885581Z","iopub.status.idle":"2023-07-28T17:08:23.889871Z","shell.execute_reply.started":"2023-07-28T17:08:23.885545Z","shell.execute_reply":"2023-07-28T17:08:23.888804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import torch\n# import os\n# from PIL import Image\n# import torch\n# from sklearn.metrics.pairwise import cosine_similarity\n\n\n# def compute_scores(emb_one, emb_two):\n#     \"\"\"Computes cosine similarity between two tensors.\"\"\"\n#     scores = torch.nn.functional.cosine_similarity(emb_one, emb_two)\n#     return scores.numpy().tolist()\n\n# import torch\n\n\n\n\n\n\n\n# import torch\n\n# def fetch_similar(image, top_k=5):\n#     \"\"\"Fetches the `top_k` similar images with `image` as the query.\"\"\"\n#     # Prepare the input query image for embedding computation.\n#     image_path = os.path.join(image_dir, image)\n#     image = Image.open(image_path)\n#     image_transformed = transformation_chain(image).unsqueeze(0)\n#     new_batch = {\"pixel_values\": image_transformed.to(device)}\n\n#     # Compute the embedding for the query image.\n#     with torch.no_grad():\n#         query_embeddings = model(**new_batch).last_hidden_state[:, 0].cpu()\n\n#     # Convert candidate embeddings DataFrame to a list of Tensors.\n#     candidate_embeddings_list = list(candidate_subset[\"embeddings\"].values)\n#     candidate_embeddings_tensor = torch.stack(candidate_embeddings_list)\n\n#     # Compute similarity scores with all the candidate images at once.\n#     sim_scores = compute_scores(candidate_embeddings_tensor, query_embeddings)\n#     similarity_mapping = dict(zip(candidate_subset[\"image_id\"], sim_scores))\n\n#     # Sort the mapping dictionary and return `top_k` candidates.\n#     similarity_mapping_sorted = dict(\n#         sorted(similarity_mapping.items(), key=lambda x: x[1], reverse=True)\n#     )\n#     id_entries = list(similarity_mapping_sorted.keys())[:top_k]\n\n#     # Ensure `id_entries` is a list of strings\n#     id_entries = [str(entry) for entry in id_entries]\n\n#     # Extract ids and labels directly from the sorted keys\n#     ids = list(map(lambda x: int(x.split(\"_\")[0]), id_entries))\n#     labels = list(map(lambda x: int(x.split(\"_\")[-1]), id_entries))\n#     return ids, labels\n# #     return query_embeddings\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:24.398806Z","iopub.execute_input":"2023-07-28T17:08:24.399210Z","iopub.status.idle":"2023-07-28T17:08:24.405439Z","shell.execute_reply.started":"2023-07-28T17:08:24.399177Z","shell.execute_reply":"2023-07-28T17:08:24.404527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fetch_similar(test_sample_file_name)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:25.624037Z","iopub.execute_input":"2023-07-28T17:08:25.624456Z","iopub.status.idle":"2023-07-28T17:08:25.629520Z","shell.execute_reply.started":"2023-07-28T17:08:25.624420Z","shell.execute_reply":"2023-07-28T17:08:25.628350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# import pandas as pd\n# from PIL import Image\n# image_dir = \"/kaggle/input/dlsprint2/badlad/images/test\"\n# # Assuming you have the DataFrame named 'test_metadata' with the necessary image data\n\n# # Randomly select a test sample index\n# test_idx = np.random.choice(len(test_metadata))\n\n# # Get the test sample image file name and other information from the DataFrame\n# test_sample_file_name = test_metadata.iloc[test_idx][\"file_name\"]\n# test_image_id = test_metadata.iloc[test_idx][\"image_id\"]\n\n# # Load the image using PIL\n# image_path = \"/kaggle/input/dlsprint2/badlad/images/test/\"  # Replace with the path to your image directory\n# query_image = Image.open(image_path + test_sample_file_name)\n\n# # Save the query image to a temporary file\n# temp_image_path = \"/tmp/query_image.png\"\n# query_image.save(temp_image_path)\n\n# # # Specify the image viewer to use (e.g., eog for Linux, Preview for macOS)\n# # image_viewer = \"eog\"  # Replace with the appropriate viewer for your system\n\n# # # Open the image using the specified viewer\n# # os.system(f\"{image_viewer} {temp_image_path}\")\n\n# # # Print the query image ID\n# # print(f\"Query image ID: {test_image_id}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:26.094408Z","iopub.execute_input":"2023-07-28T17:08:26.094782Z","iopub.status.idle":"2023-07-28T17:08:26.100783Z","shell.execute_reply.started":"2023-07-28T17:08:26.094753Z","shell.execute_reply":"2023-07-28T17:08:26.099601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_idx = np.random.choice(len(test_metadata))\n# test_sample_file_name = test_metadata.iloc[test_idx][\"file_name\"]\n# train_idx = np.random.choice(len(train_metadata))\n# train_sample_file_name = train_metadata.iloc[train_idx][\"file_name\"]\n# test_image_id = test_metadata.iloc[test_idx][\"image_id\"]\n\n# sim_ids, file_name = fetch_similar(test_sample_file_name)\n\n# print(f\"Query label: {train_sample_file_name}\")\n# print(f\"Top 5 candidate labels: {sim_ids}\")","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:27.003049Z","iopub.execute_input":"2023-07-28T17:08:27.003462Z","iopub.status.idle":"2023-07-28T17:08:27.008477Z","shell.execute_reply.started":"2023-07-28T17:08:27.003427Z","shell.execute_reply":"2023-07-28T17:08:27.007334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Example filename\n\n# # Get the full path to the image file\n# image_path = \"/kaggle/input/dlsprint2/badlad/images/train/\" +train_sample_file_name\n\n# # Open the image using PIL\n# image = Image.open(image_path)\n# # Display the image using matplotlib\n# plt.imshow(image)\n# plt.axis('off')  # Turn off axis labels\n# plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:27.553454Z","iopub.execute_input":"2023-07-28T17:08:27.553849Z","iopub.status.idle":"2023-07-28T17:08:27.558958Z","shell.execute_reply.started":"2023-07-28T17:08:27.553817Z","shell.execute_reply":"2023-07-28T17:08:27.557464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# associated_file_names = []\n\n\n# for image_id in sim_ids:\n#     # Use loc to find the row with the matching image_id and extract the file_name\n#     file_name = candidate_subset.loc[candidate_subset['image_id'] == image_id, 'file_name'].values[0]\n#     associated_file_names.append(file_name)\n\n# # Print the associated file names\n# print(associated_file_names)\n\n# image_dir = \"/kaggle/input/dlsprint2/badlad/images/test\"  # Replace with the actual path to your image directory\n\n# # Function to display images\n# def display_image(image_path):\n#     image = Image.open(image_path)\n#     plt.imshow(image)\n#     plt.axis('off')\n#     plt.show()\n\n# # Visualize each image\n# for file_name in associated_file_names:\n#     image_path = os.path.join(image_dir, file_name)\n#     display_image(image_path)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T17:08:28.045019Z","iopub.execute_input":"2023-07-28T17:08:28.045456Z","iopub.status.idle":"2023-07-28T17:08:28.051255Z","shell.execute_reply.started":"2023-07-28T17:08:28.045412Z","shell.execute_reply":"2023-07-28T17:08:28.049800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"**new testing ****","metadata":{}},{"cell_type":"code","source":"# # Initialize your model and transformation_chain as needed\n\n# # Set the device for the model\n# device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n# model = model.to(device)\n# image_dir = \"/kaggle/input/dlsprint2/badlad/images/test\"\n\n# # Create a new empty column for embeddings in the DataFrame\n# candidate_subset[\"embeddings\"] = None\n\n# # Define the function to compute embeddings for a single image\n# def compute_embeddings(image_filename):\n#     image_path = os.path.join(image_dir, image_filename)\n#     image = Image.open(image_path)\n#     image=image.convert(\"RGB\")\n#     image_transformed = transformation_chain(image)\n#     new_batch = {\"pixel_values\": image_transformed.unsqueeze(0).to(device)}\n#     with torch.no_grad():\n#         embeddings = model(**new_batch).last_hidden_state[:, 0].cpu()\n#         return embeddings.tolist()\n# #     return embeddings[0].tolist()\n\n# # Apply the compute_embeddings function to each image in the DataFrame\n# candidate_subset[\"embeddings\"] = candidate_subset[\"file_name\"].apply(compute_embeddings)\n\n# # Now 'candidate_subset' contains the embeddings for each image\n# print(candidate_subset)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(image_df)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:44:44.043147Z","iopub.execute_input":"2023-08-02T07:44:44.044098Z","iopub.status.idle":"2023-08-02T07:44:44.054714Z","shell.execute_reply.started":"2023-08-02T07:44:44.044034Z","shell.execute_reply":"2023-08-02T07:44:44.053227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# Assuming you have the DataFrame with image data called 'dataset'\n# dataset = pd.read_csv('your_file.csv')\n\n# Define the number of samples you want to select\nnum_samples = 18000\n\n# Define the random seed for reproducibility\nseed = 42\n\n# Shuffle the DataFrame using the random seed\nshuffled_dataset = image_df.sample(frac=1, random_state=seed)\n\n# Select the first 'num_samples' rows from the shuffled DataFrame\ncandidate_subset1 = shuffled_dataset.head(num_samples)\n\n# Now 'candidate_subset' contains the randomly selected 100 images\nprint(candidate_subset1)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:44:50.606903Z","iopub.execute_input":"2023-08-02T07:44:50.607293Z","iopub.status.idle":"2023-08-02T07:44:50.629995Z","shell.execute_reply.started":"2023-08-02T07:44:50.607263Z","shell.execute_reply":"2023-08-02T07:44:50.627954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torchvision.transforms as T\n\n\n# Data transformation chain.\ntransformation_chain = T.Compose(\n    [\n        # We first resize the input image to 256x256 and then we take center crop.\n        T.Resize(int((256 / 224) * extractor.size[\"height\"])),\n        T.CenterCrop(extractor.size[\"height\"]),\n        T.ToTensor(),\n        T.Normalize(mean=extractor.image_mean, std=extractor.image_std),\n    ]\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:44:59.955350Z","iopub.execute_input":"2023-08-02T07:44:59.955805Z","iopub.status.idle":"2023-08-02T07:45:00.370581Z","shell.execute_reply.started":"2023-08-02T07:44:59.955769Z","shell.execute_reply":"2023-08-02T07:45:00.369443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Initialize your model and transformation_chain as needed\n\n# Set the device for the model\ndevice = \"cuda\" if torch.cuda.is_available() else \"cpu\"\nmodel = model.to(device)\nimage_dir = \"/kaggle/input/badlad-inference-pseudolabels-p6/epaper\"\n\n# Create a new empty column for embeddings in the DataFrame\ncandidate_subset1[\"embeddings\"] = None\n\n# Define the function to compute embeddings for a single image\ndef compute_embeddings(image_filename):\n    image_path = os.path.join(image_dir, image_filename)\n    image = Image.open(image_path)\n    image=image.convert(\"RGB\")\n    image_transformed = transformation_chain(image)\n    new_batch = {\"pixel_values\": image_transformed.unsqueeze(0).to(device)}\n    with torch.no_grad():\n        embeddings = model(**new_batch).last_hidden_state[:, 0].cpu()\n        return embeddings.tolist()\n#     return embeddings[0].tolist()\n\n# Apply the compute_embeddings function to each image in the DataFrame\ncandidate_subset1[\"embeddings\"] = candidate_subset1[\"file_name\"].apply(compute_embeddings)\n\n# Now 'candidate_subset' contains the embeddings for each image\nprint(candidate_subset1)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T07:45:44.822761Z","iopub.execute_input":"2023-08-02T07:45:44.823190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(candidate_subset1[\"embeddings\"])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"two_dimensional_list = candidate_subset1[\"embeddings\"].tolist()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:15:42.789798Z","iopub.execute_input":"2023-07-28T13:15:42.791164Z","iopub.status.idle":"2023-07-28T13:15:42.798245Z","shell.execute_reply.started":"2023-07-28T13:15:42.791117Z","shell.execute_reply":"2023-07-28T13:15:42.796863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport torch\n\n# Assuming 'df' is your DataFrame with the \"embeddings\" column\n# The \"embeddings\" column contains lists of lists in each row\n\n# Convert the \"embeddings\" column to a tensor with the desired shape\ncandidate_subset1[\"embeddings\"] = candidate_subset1[\"embeddings\"].apply(lambda x: torch.tensor(x, dtype=torch.float64))\n\n# Now 'df[\"embeddings\"]' contains tensors with the desired shape\nprint(candidate_subset1[\"embeddings\"])","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:15:44.269516Z","iopub.execute_input":"2023-07-28T13:15:44.269941Z","iopub.status.idle":"2023-07-28T13:15:46.621333Z","shell.execute_reply.started":"2023-07-28T13:15:44.269907Z","shell.execute_reply":"2023-07-28T13:15:46.620079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\nall_candidate_embeddings = np.array(two_dimensional_list)\nall_candidate_embeddings = torch.from_numpy(all_candidate_embeddings)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:15:49.060771Z","iopub.execute_input":"2023-07-28T13:15:49.061971Z","iopub.status.idle":"2023-07-28T13:15:50.690376Z","shell.execute_reply.started":"2023-07-28T13:15:49.061926Z","shell.execute_reply":"2023-07-28T13:15:50.689206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport torch\n\n# Assuming 'all_candidate_embeddings' is your tensor with shape (100, 1, 768)\n\n# Reshape the tensor to have shape (100, 768)\nreshaped_embeddings = all_candidate_embeddings.view(18000, 768)\n\n# Convert the reshaped tensor to a pandas DataFrame\ncolumn_names = [f\"embedding_{i}\" for i in range(reshaped_embeddings.shape[1])]\nall_candidate_embeddings_df = pd.DataFrame(reshaped_embeddings, columns=column_names)\n\n# Now 'all_candidate_embeddings_df' is a DataFrame with the desired shape\nprint(all_candidate_embeddings_df)\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:16:01.099028Z","iopub.execute_input":"2023-07-28T13:16:01.099490Z","iopub.status.idle":"2023-07-28T13:16:01.131144Z","shell.execute_reply.started":"2023-07-28T13:16:01.099453Z","shell.execute_reply":"2023-07-28T13:16:01.130117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport os\nfrom PIL import Image\nimport torch\nfrom sklearn.metrics.pairwise import cosine_similarity\n\n\ndef compute_scores(emb_one, emb_two):\n    \"\"\"Computes cosine similarity between two tensors.\"\"\"\n    scores = torch.nn.functional.cosine_similarity(emb_one, emb_two)\n    return scores.numpy().tolist()\n\nimport torch\n\ndef fetch_similar(image, top_k=5):\n    \"\"\"Fetches the `top_k` similar images with `image` as the query.\"\"\"\n    # Prepare the input query image for embedding computation.\n    image_path = os.path.join(image_dir, image)\n    image = Image.open(image_path)\n    image_transformed = transformation_chain(image).unsqueeze(0)\n    new_batch = {\"pixel_values\": image_transformed.to(device)}\n\n    # Compute the embedding for the query image.\n    with torch.no_grad():\n        query_embeddings = model(**new_batch).last_hidden_state[:, 0].cpu()\n\n    # Convert candidate embeddings DataFrame to a list of Tensors.\n    candidate_embeddings_list = list(candidate_subset1[\"embeddings\"].values)\n    candidate_embeddings_tensor = torch.stack(candidate_embeddings_list)\n\n    # Compute similarity scores with all the candidate images at once.\n    sim_scores = compute_scores(candidate_embeddings_tensor, query_embeddings)\n    similarity_mapping = dict(zip(candidate_subset1[\"image_id\"], sim_scores))\n\n    # Sort the mapping dictionary and return `top_k` candidates.\n    similarity_mapping_sorted = dict(\n        sorted(similarity_mapping.items(), key=lambda x: x[1], reverse=True)\n    )\n    id_entries = list(similarity_mapping_sorted.keys())[:top_k]\n\n    # Ensure `id_entries` is a list of strings\n    id_entries = [str(entry) for entry in id_entries]\n\n    # Extract ids and labels directly from the sorted keys\n    ids = list(map(lambda x: int(x.split(\"_\")[0]), id_entries))\n    labels = list(map(lambda x: int(x.split(\"_\")[-1]), id_entries))\n    return ids, labels\n#     return query_embeddings","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:26:35.651813Z","iopub.execute_input":"2023-07-28T13:26:35.652302Z","iopub.status.idle":"2023-07-28T13:26:35.667093Z","shell.execute_reply.started":"2023-07-28T13:26:35.652267Z","shell.execute_reply":"2023-07-28T13:26:35.665856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"candidate_idx = np.random.choice(len(image_df))\ncandidate_sample_file_name = image_df.iloc[candidate_idx][\"file_name\"]\ntest_idx = np.random.choice(len(test_metadata))\ntest_sample_file_name = test_metadata.iloc[test_idx][\"file_name\"]\ntest_image_id = test_metadata.iloc[test_idx][\"image_id\"]\n\nsim_ids, file_name = fetch_similar(candidate_sample_file_name)\n\nprint(f\"Query label: {candidate_sample_file_name}\")\nprint(f\"Top 5 candidate labels: {sim_ids}\")","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:29:23.980325Z","iopub.execute_input":"2023-07-28T13:29:23.981702Z","iopub.status.idle":"2023-07-28T13:29:26.813538Z","shell.execute_reply.started":"2023-07-28T13:29:23.981641Z","shell.execute_reply":"2023-07-28T13:29:26.812615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example filename\n\n# Get the full path to the image file\nimage_path = \"/kaggle/input/dlsprint2/badlad/images/test/\" +test_sample_file_name\n\n# Open the image using PIL\nimage = Image.open(image_path)\n# Display the image using matplotlib\nplt.imshow(image)\nplt.axis('off')  # Turn off axis labels\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:29:28.145996Z","iopub.execute_input":"2023-07-28T13:29:28.147148Z","iopub.status.idle":"2023-07-28T13:29:28.482289Z","shell.execute_reply.started":"2023-07-28T13:29:28.147108Z","shell.execute_reply":"2023-07-28T13:29:28.481001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"associated_file_names = []\n\n\nfor image_id in sim_ids:\n    # Use loc to find the row with the matching image_id and extract the file_name\n    file_name = candidate_subset1.loc[candidate_subset1['image_id'] == image_id, 'file_name'].values[0]\n    associated_file_names.append(file_name)\n\n# Print the associated file names\nprint(associated_file_names)\n\nimage_dir = \"/kaggle/input/badlad-inference-pseudolabels/zipfile10\"  # Replace with the actual path to your image directory\n\n# Function to display images\ndef display_image(image_path):\n    image = Image.open(image_path)\n    plt.imshow(image)\n    plt.axis('off')\n    plt.show()\n\n# Visualize each image\nfor file_name in associated_file_names:\n    image_path = os.path.join(image_dir, file_name)\n    display_image(image_path)","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:29:30.055782Z","iopub.execute_input":"2023-07-28T13:29:30.056198Z","iopub.status.idle":"2023-07-28T13:29:32.912176Z","shell.execute_reply.started":"2023-07-28T13:29:30.056167Z","shell.execute_reply":"2023-07-28T13:29:32.910814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**use loop to generate query image and it's corresponding candidate similar image and download it in a fodler**","metadata":{}},{"cell_type":"code","source":"import os\nimport shutil\n\n# Define source and target folder paths\nsource_folder = \"/kaggle/input/badlad-inference-pseudolabels/zipfile10/\"\ntarget_folder = \"/kaggle/working/newdata1\"\n\n# Create the target folder if it doesn't exist\nos.makedirs(target_folder, exist_ok=True)\n\n# Loop to generate query labels and retrieve similar images for test_idx\nfor _ in range(200):  # Replace 2 with the number of times you want to repeat the process for test_idx\n    # Generate random test_idx and get the test sample file name and image_id\n    candidate_idx = np.random.choice(len(image_df))\n    candidate_sample_file_name = image_df.iloc[candidate_idx][\"file_name\"]\n    test_idx = np.random.choice(len(test_metadata))\n    test_sample_file_name = test_metadata.iloc[test_idx][\"file_name\"]\n    test_image_id = test_metadata.iloc[test_idx][\"image_id\"]\n\n    sim_ids, file_name = fetch_similar(candidate_sample_file_name)\n\n    print(f\"Query label: {test_sample_file_name}\")\n    print(f\"Top 5 candidate labels: {sim_ids}\")\n\n    associated_file_names = []\n\n    for image_id in sim_ids:\n        # Use loc to find the row with the matching image_id and extract the file_name\n        file_name = candidate_subset1.loc[candidate_subset1['image_id'] == image_id, 'file_name'].values[0]\n        associated_file_names.append(file_name)\n\n    # Copy the matching images to the target folder\n    for filename in os.listdir(source_folder):\n        if filename in associated_file_names:\n            source_path = os.path.join(source_folder, filename)\n            target_path = os.path.join(target_folder, filename)\n            shutil.copyfile(source_path, target_path)\n    # Example filename\n#     print(\"query image\")\n    # Get the full path to the image file\n#     image_path = \"/kaggle/input/dlsprint2/badlad/images/test/\" +test_sample_file_name\n\n#     # Open the image using PIL\n#     image = Image.open(image_path)\n#     # Display the image using matplotlib\n#     plt.imshow(image)\n#     plt.axis('off')  # Turn off axis labels\n#     plt.show()\n    \n#     print(\"candidate images\")\n\n            \n#     def display_image(image_path):\n#         image = Image.open(image_path)\n#         plt.imshow(image)\n#         plt.axis('off')\n#         plt.show()\n\n#     # Visualize each image\n#     for file_name in associated_file_names:\n#         image_path = os.path.join(image_dir, file_name)\n#         display_image(image_path)\n\nprint(\"*****finished*****\")","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:35:06.944675Z","iopub.execute_input":"2023-07-28T13:35:06.945207Z","iopub.status.idle":"2023-07-28T13:44:50.894126Z","shell.execute_reply.started":"2023-07-28T13:35:06.945170Z","shell.execute_reply":"2023-07-28T13:44:50.893011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**make a zip file of the folder**","metadata":{}},{"cell_type":"code","source":"import zipfile\nimport os\n\n# Path to the folder you want to zip\nfolder_path = \"/kaggle/working/newdata1\"\n\n# Path and name of the zip file to create\nzip_file_path = \"/kaggle/working/newdata_unanno.zip\"\n\n# Create a zip file and write the contents of the folder to it\nwith zipfile.ZipFile(zip_file_path, 'w') as zipf:\n    for root, _, files in os.walk(folder_path):\n        for file in files:\n            file_path = os.path.join(root, file)\n            zipf.write(file_path, os.path.relpath(file_path, folder_path))\n\n# Once the zip file is created, you can download it from the Kaggle output folder\n","metadata":{"execution":{"iopub.status.busy":"2023-07-28T13:51:06.157147Z","iopub.execute_input":"2023-07-28T13:51:06.157871Z","iopub.status.idle":"2023-07-28T13:51:12.927784Z","shell.execute_reply.started":"2023-07-28T13:51:06.157823Z","shell.execute_reply":"2023-07-28T13:51:12.926674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}